{"timestamp_utc": "2026-04-12T22:14:34Z", "mode": "train", "global_step": 1, "epoch": 0.00010045203415369161, "loss": 0.1289, "grad_norm": 24.25541877746582, "learning_rate": 1e-05, "num_tokens": 1670.0, "completions/mean_length": 41.75, "completions/min_length": 29.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.75, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.6725290417671204, "rewards/meter/std": 0.40340036153793335, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9957022070884705, "rewards/repeat_soft/std": 0.00409209867939353, "rewards/judge_quality/mean": 0.4762499928474426, "rewards/judge_quality/std": 0.09941796213388443, "rewards/total_composite/mean": 0.6763333082199097, "rewards/total_composite/std": 0.20289066433906555, "reward": 0.6763333082199097, "reward_std": 0.20289064943790436, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22603146731853485, "sampling/sampling_logp_difference/max": 1.5626206398010254, "sampling/importance_sampling_ratio/min": 0.20958609879016876, "sampling/importance_sampling_ratio/mean": 1.013262391090393, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.01923768222332, "clip_ratio/low_mean": 0.08834622986614704, "clip_ratio/low_min": 0.08834622986614704, "clip_ratio/high_mean": 0.15415428206324577, "clip_ratio/high_max": 0.15415428206324577, "clip_ratio/region_mean": 0.24250051192939281, "reward_total_mean": 0.6763333082199097, "reward_meter_mean": 0.6725290417671204, "reward_meter_std": 0.40340036153793335, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9957022070884705, "reward_repeat_soft_std": 0.00409209867939353, "reward_judge_quality_mean": 0.4762499928474426, "reward_judge_quality_std": 0.09941796213388443, "reward_total_composite_mean": 0.6763333082199097, "reward_total_composite_std": 0.20289066433906555} {"timestamp_utc": "2026-04-12T22:14:41Z", "mode": "train", "global_step": 2, "epoch": 0.00020090406830738323, "loss": 0.0733, "grad_norm": 9.701464653015137, "learning_rate": 9.996969696969698e-06, "num_tokens": 4111.0, "completions/mean_length": 141.125, "completions/min_length": 106.0, "completions/max_length": 161.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 141.125, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 161.0, "rewards/meter/mean": 0.667778730392456, "rewards/meter/std": 0.3386266231536865, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9944028854370117, "rewards/repeat_soft/std": 0.00460322480648756, "rewards/judge_quality/mean": 0.6575000286102295, "rewards/judge_quality/std": 0.2133910059928894, "rewards/total_composite/mean": 0.7471907138824463, "rewards/total_composite/std": 0.150190070271492, "reward": 0.7471907138824463, "reward_std": 0.1501900553703308, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2034754604101181, "sampling/sampling_logp_difference/max": 1.743311882019043, "sampling/importance_sampling_ratio/min": 0.19116829335689545, "sampling/importance_sampling_ratio/mean": 1.0195162296295166, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9704072922468185, "clip_ratio/low_mean": 0.05871324986219406, "clip_ratio/low_min": 0.05871324986219406, "clip_ratio/high_mean": 0.128573689609766, "clip_ratio/high_max": 0.128573689609766, "clip_ratio/region_mean": 0.18728693947196007, "reward_total_mean": 0.7471907138824463, "reward_meter_mean": 0.667778730392456, "reward_meter_std": 0.3386266231536865, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9944028854370117, "reward_repeat_soft_std": 0.00460322480648756, "reward_judge_quality_mean": 0.6575000286102295, "reward_judge_quality_std": 0.2133910059928894, "reward_total_composite_mean": 0.7471907138824463, "reward_total_composite_std": 0.150190070271492} {"timestamp_utc": "2026-04-12T22:14:47Z", "mode": "train", "global_step": 3, "epoch": 0.00030135610246107485, "loss": 0.0512, "grad_norm": 21.948942184448242, "learning_rate": 9.993939393939395e-06, "num_tokens": 5819.0, "completions/mean_length": 36.5, "completions/min_length": 24.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.5, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.34533801674842834, "rewards/meter/std": 0.3437061011791229, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9806680679321289, "rewards/repeat_soft/std": 0.02873978205025196, "rewards/judge_quality/mean": 0.6112500429153442, "rewards/judge_quality/std": 0.25587037205696106, "rewards/total_composite/mean": 0.5868439078330994, "rewards/total_composite/std": 0.20601670444011688, "reward": 0.5868439078330994, "reward_std": 0.20601673424243927, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.319444864988327, "sampling/sampling_logp_difference/max": 2.2889962196350098, "sampling/importance_sampling_ratio/min": 0.10136815905570984, "sampling/importance_sampling_ratio/mean": 0.9693852066993713, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3289805501699448, "clip_ratio/low_mean": 0.15689394064247608, "clip_ratio/low_min": 0.15689394064247608, "clip_ratio/high_mean": 0.07748956605792046, "clip_ratio/high_max": 0.07748956605792046, "clip_ratio/region_mean": 0.23438350670039654, "reward_total_mean": 0.5868439078330994, "reward_meter_mean": 0.34533801674842834, "reward_meter_std": 0.3437061011791229, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9806680679321289, "reward_repeat_soft_std": 0.02873978205025196, "reward_judge_quality_mean": 0.6112500429153442, "reward_judge_quality_std": 0.25587037205696106, "reward_total_composite_mean": 0.5868439078330994, "reward_total_composite_std": 0.20601670444011688} {"timestamp_utc": "2026-04-12T22:14:53Z", "mode": "train", "global_step": 4, "epoch": 0.00040180813661476645, "loss": 0.0029, "grad_norm": 17.112354278564453, "learning_rate": 9.990909090909093e-06, "num_tokens": 7420.0, "completions/mean_length": 43.125, "completions/min_length": 35.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.125, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.6681925058364868, "rewards/meter/std": 0.34428924322128296, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9990890026092529, "rewards/repeat_soft/std": 0.00257677398622036, "rewards/judge_quality/mean": 0.7150000333786011, "rewards/judge_quality/std": 0.2377273440361023, "rewards/total_composite/mean": 0.7650955319404602, "rewards/total_composite/std": 0.1345381885766983, "reward": 0.7650955319404602, "reward_std": 0.1345381736755371, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21286356449127197, "sampling/sampling_logp_difference/max": 1.6740360260009766, "sampling/importance_sampling_ratio/min": 0.18748880922794342, "sampling/importance_sampling_ratio/mean": 1.0222951173782349, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9584360122680664, "clip_ratio/low_mean": 0.08466819114983082, "clip_ratio/low_min": 0.08466819114983082, "clip_ratio/high_mean": 0.13993905391544104, "clip_ratio/high_max": 0.13993905391544104, "clip_ratio/region_mean": 0.22460724506527185, "reward_total_mean": 0.7650955319404602, "reward_meter_mean": 0.6681925058364868, "reward_meter_std": 0.34428924322128296, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9990890026092529, "reward_repeat_soft_std": 0.00257677398622036, "reward_judge_quality_mean": 0.7150000333786011, "reward_judge_quality_std": 0.2377273440361023, "reward_total_composite_mean": 0.7650955319404602, "reward_total_composite_std": 0.1345381885766983} {"timestamp_utc": "2026-04-12T22:15:00Z", "mode": "train", "global_step": 5, "epoch": 0.0005022601707684581, "loss": 0.0895, "grad_norm": 9.455750465393066, "learning_rate": 9.987878787878788e-06, "num_tokens": 9686.0, "completions/mean_length": 106.25, "completions/min_length": 71.0, "completions/max_length": 133.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.25, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.5534956455230713, "rewards/meter/std": 0.35001081228256226, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9964775443077087, "rewards/repeat_soft/std": 0.0026533447671681643, "rewards/judge_quality/mean": 0.7099999785423279, "rewards/judge_quality/std": 0.24213339388370514, "rewards/total_composite/mean": 0.7070332765579224, "rewards/total_composite/std": 0.182778000831604, "reward": 0.7070332765579224, "reward_std": 0.1827780157327652, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19017860293388367, "sampling/sampling_logp_difference/max": 1.9287948608398438, "sampling/importance_sampling_ratio/min": 0.14532321691513062, "sampling/importance_sampling_ratio/mean": 1.021316409111023, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3343858867883682, "clip_ratio/low_mean": 0.0882453341037035, "clip_ratio/low_min": 0.0882453341037035, "clip_ratio/high_mean": 0.0963182132691145, "clip_ratio/high_max": 0.0963182132691145, "clip_ratio/region_mean": 0.184563547372818, "reward_total_mean": 0.7070332765579224, "reward_meter_mean": 0.5534956455230713, "reward_meter_std": 0.35001081228256226, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9964775443077087, "reward_repeat_soft_std": 0.0026533447671681643, "reward_judge_quality_mean": 0.7099999785423279, "reward_judge_quality_std": 0.24213339388370514, "reward_total_composite_mean": 0.7070332765579224, "reward_total_composite_std": 0.182778000831604} {"timestamp_utc": "2026-04-12T22:15:07Z", "mode": "train", "global_step": 6, "epoch": 0.0006027122049221497, "loss": 0.1401, "grad_norm": 9.252427101135254, "learning_rate": 9.984848484848485e-06, "num_tokens": 12284.0, "completions/mean_length": 125.75, "completions/min_length": 85.0, "completions/max_length": 158.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 125.75, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 158.0, "rewards/meter/mean": 0.7442054748535156, "rewards/meter/std": 0.36003756523132324, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9989632368087769, "rewards/repeat_soft/std": 0.0006836142274551094, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.7295387983322144, "rewards/total_composite/std": 0.1836954951286316, "reward": 0.7295387983322144, "reward_std": 0.1836954802274704, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20029257237911224, "sampling/sampling_logp_difference/max": 1.7579917907714844, "sampling/importance_sampling_ratio/min": 0.17239071428775787, "sampling/importance_sampling_ratio/mean": 1.0128437280654907, "sampling/importance_sampling_ratio/max": 1.9965848922729492, "entropy": 2.4172347486019135, "clip_ratio/low_mean": 0.07391603104770184, "clip_ratio/low_min": 0.07391603104770184, "clip_ratio/high_mean": 0.1392985973507166, "clip_ratio/high_max": 0.1392985973507166, "clip_ratio/region_mean": 0.21321462839841843, "reward_total_mean": 0.7295387983322144, "reward_meter_mean": 0.7442054748535156, "reward_meter_std": 0.36003756523132324, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9989632368087769, "reward_repeat_soft_std": 0.0006836142274551094, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.7295387983322144, "reward_total_composite_std": 0.1836954951286316} {"timestamp_utc": "2026-04-12T22:15:14Z", "mode": "train", "global_step": 7, "epoch": 0.0007031642390758413, "loss": -0.0576, "grad_norm": 13.074331283569336, "learning_rate": 9.981818181818183e-06, "num_tokens": 14661.0, "completions/mean_length": 113.125, "completions/min_length": 84.0, "completions/max_length": 159.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 113.125, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 159.0, "rewards/meter/mean": 0.7239573001861572, "rewards/meter/std": 0.26627305150032043, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9978160858154297, "rewards/repeat_soft/std": 0.001829512882977724, "rewards/judge_quality/mean": 0.5325000286102295, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.7315623760223389, "rewards/total_composite/std": 0.12214145064353943, "reward": 0.7315623760223389, "reward_std": 0.12214145064353943, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2251056730747223, "sampling/sampling_logp_difference/max": 1.424403190612793, "sampling/importance_sampling_ratio/min": 0.24065205454826355, "sampling/importance_sampling_ratio/mean": 1.03545081615448, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.566889613866806, "clip_ratio/low_mean": 0.07710661552846432, "clip_ratio/low_min": 0.07710661552846432, "clip_ratio/high_mean": 0.1532231941819191, "clip_ratio/high_max": 0.1532231941819191, "clip_ratio/region_mean": 0.23032980971038342, "reward_total_mean": 0.7315623760223389, "reward_meter_mean": 0.7239573001861572, "reward_meter_std": 0.26627305150032043, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9978160858154297, "reward_repeat_soft_std": 0.001829512882977724, "reward_judge_quality_mean": 0.5325000286102295, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.7315623760223389, "reward_total_composite_std": 0.12214145064353943} {"timestamp_utc": "2026-04-12T22:15:20Z", "mode": "train", "global_step": 8, "epoch": 0.0008036162732295329, "loss": 0.0692, "grad_norm": 37.71578598022461, "learning_rate": 9.97878787878788e-06, "num_tokens": 16079.0, "completions/mean_length": 21.25, "completions/min_length": 16.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.25, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.8553460836410522, "rewards/meter/std": 0.32826167345046997, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9081118702888489, "rewards/repeat_soft/std": 0.10070746392011642, "rewards/judge_quality/mean": 0.65625, "rewards/judge_quality/std": 0.29621124267578125, "rewards/total_composite/mean": 0.8225919008255005, "rewards/total_composite/std": 0.1914392113685608, "reward": 0.8225919008255005, "reward_std": 0.1914391964673996, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20113249123096466, "sampling/sampling_logp_difference/max": 1.2836437225341797, "sampling/importance_sampling_ratio/min": 0.2770260274410248, "sampling/importance_sampling_ratio/mean": 1.0172845125198364, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6074406504631042, "clip_ratio/low_mean": 0.06774115934967995, "clip_ratio/low_min": 0.06774115934967995, "clip_ratio/high_mean": 0.07261904887855053, "clip_ratio/high_max": 0.07261904887855053, "clip_ratio/region_mean": 0.14036020822823048, "reward_total_mean": 0.8225919008255005, "reward_meter_mean": 0.8553460836410522, "reward_meter_std": 0.32826167345046997, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9081118702888489, "reward_repeat_soft_std": 0.10070746392011642, "reward_judge_quality_mean": 0.65625, "reward_judge_quality_std": 0.29621124267578125, "reward_total_composite_mean": 0.8225919008255005, "reward_total_composite_std": 0.1914392113685608} {"timestamp_utc": "2026-04-12T22:15:27Z", "mode": "train", "global_step": 9, "epoch": 0.0009040683073832245, "loss": 0.0029, "grad_norm": 9.873559951782227, "learning_rate": 9.975757575757577e-06, "num_tokens": 18924.0, "completions/mean_length": 149.625, "completions/min_length": 105.0, "completions/max_length": 190.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 149.625, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 190.0, "rewards/meter/mean": 0.8943005800247192, "rewards/meter/std": 0.12937459349632263, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9977504014968872, "rewards/repeat_soft/std": 0.0031607486307621002, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.713111937046051, "rewards/total_composite/std": 0.29207462072372437, "reward": 0.713111937046051, "reward_std": 0.29207465052604675, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21067966520786285, "sampling/sampling_logp_difference/max": 1.7020273208618164, "sampling/importance_sampling_ratio/min": 0.18231353163719177, "sampling/importance_sampling_ratio/mean": 1.0195698738098145, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.251196339726448, "clip_ratio/low_mean": 0.01953125, "clip_ratio/low_min": 0.01953125, "clip_ratio/high_mean": 0.2094412874430418, "clip_ratio/high_max": 0.2094412874430418, "clip_ratio/region_mean": 0.2289725374430418, "reward_total_mean": 0.713111937046051, "reward_meter_mean": 0.8943005800247192, "reward_meter_std": 0.12937459349632263, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9977504014968872, "reward_repeat_soft_std": 0.0031607486307621002, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.713111937046051, "reward_total_composite_std": 0.29207462072372437} {"timestamp_utc": "2026-04-12T22:15:34Z", "mode": "train", "global_step": 10, "epoch": 0.0010045203415369162, "loss": 0.0833, "grad_norm": 25.13707733154297, "learning_rate": 9.972727272727274e-06, "num_tokens": 20493.0, "completions/mean_length": 34.125, "completions/min_length": 25.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.125, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.6848095059394836, "rewards/meter/std": 0.4096945524215698, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9840165376663208, "rewards/repeat_soft/std": 0.029735958203673363, "rewards/judge_quality/mean": 0.7325000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.6806565523147583, "rewards/total_composite/std": 0.34453243017196655, "reward": 0.6806565523147583, "reward_std": 0.34453243017196655, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21705979108810425, "sampling/sampling_logp_difference/max": 1.3225932121276855, "sampling/importance_sampling_ratio/min": 0.2664434611797333, "sampling/importance_sampling_ratio/mean": 1.0199761390686035, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.476714864373207, "clip_ratio/low_mean": 0.0765726137906313, "clip_ratio/low_min": 0.0765726137906313, "clip_ratio/high_mean": 0.11742378678172827, "clip_ratio/high_max": 0.11742378678172827, "clip_ratio/region_mean": 0.19399640057235956, "reward_total_mean": 0.6806565523147583, "reward_meter_mean": 0.6848095059394836, "reward_meter_std": 0.4096945524215698, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9840165376663208, "reward_repeat_soft_std": 0.029735958203673363, "reward_judge_quality_mean": 0.7325000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.6806565523147583, "reward_total_composite_std": 0.34453243017196655} {"timestamp_utc": "2026-04-12T22:15:39Z", "mode": "train", "global_step": 11, "epoch": 0.0011049723756906078, "loss": 0.1787, "grad_norm": 18.680042266845703, "learning_rate": 9.96969696969697e-06, "num_tokens": 22069.0, "completions/mean_length": 47.0, "completions/min_length": 28.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.0, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.6532547473907471, "rewards/meter/std": 0.3094608187675476, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9933717846870422, "rewards/repeat_soft/std": 0.007688070647418499, "rewards/judge_quality/mean": 0.5824999809265137, "rewards/judge_quality/std": 0.23260943591594696, "rewards/total_composite/mean": 0.7180517911911011, "rewards/total_composite/std": 0.15922589600086212, "reward": 0.7180517911911011, "reward_std": 0.15922589600086212, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20565736293792725, "sampling/sampling_logp_difference/max": 0.9617314338684082, "sampling/importance_sampling_ratio/min": 0.3822305202484131, "sampling/importance_sampling_ratio/mean": 1.0575566291809082, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3583329617977142, "clip_ratio/low_mean": 0.09450654499232769, "clip_ratio/low_min": 0.09450654499232769, "clip_ratio/high_mean": 0.12003943882882595, "clip_ratio/high_max": 0.12003943882882595, "clip_ratio/region_mean": 0.21454598382115364, "reward_total_mean": 0.7180517911911011, "reward_meter_mean": 0.6532547473907471, "reward_meter_std": 0.3094608187675476, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9933717846870422, "reward_repeat_soft_std": 0.007688070647418499, "reward_judge_quality_mean": 0.5824999809265137, "reward_judge_quality_std": 0.23260943591594696, "reward_total_composite_mean": 0.7180517911911011, "reward_total_composite_std": 0.15922589600086212} {"timestamp_utc": "2026-04-12T22:15:45Z", "mode": "train", "global_step": 12, "epoch": 0.0012054244098442994, "loss": 0.0339, "grad_norm": 28.459325790405273, "learning_rate": 9.966666666666667e-06, "num_tokens": 23560.0, "completions/mean_length": 26.375, "completions/min_length": 16.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.375, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.6143099069595337, "rewards/meter/std": 0.47742435336112976, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9266142845153809, "rewards/repeat_soft/std": 0.07501126825809479, "rewards/judge_quality/mean": 0.39375001192092896, "rewards/judge_quality/std": 0.09941794723272324, "rewards/total_composite/mean": 0.637225866317749, "rewards/total_composite/std": 0.23131398856639862, "reward": 0.637225866317749, "reward_std": 0.23131398856639862, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2075686752796173, "sampling/sampling_logp_difference/max": 1.0368385314941406, "sampling/importance_sampling_ratio/min": 0.3545738756656647, "sampling/importance_sampling_ratio/mean": 1.0258712768554688, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.740403950214386, "clip_ratio/low_mean": 0.06502757407724857, "clip_ratio/low_min": 0.06502757407724857, "clip_ratio/high_mean": 0.10130646172910929, "clip_ratio/high_max": 0.10130646172910929, "clip_ratio/region_mean": 0.16633403580635786, "reward_total_mean": 0.637225866317749, "reward_meter_mean": 0.6143099069595337, "reward_meter_std": 0.47742435336112976, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9266142845153809, "reward_repeat_soft_std": 0.07501126825809479, "reward_judge_quality_mean": 0.39375001192092896, "reward_judge_quality_std": 0.09941794723272324, "reward_total_composite_mean": 0.637225866317749, "reward_total_composite_std": 0.23131398856639862} {"timestamp_utc": "2026-04-12T22:15:51Z", "mode": "train", "global_step": 13, "epoch": 0.001305876443997991, "loss": -0.0121, "grad_norm": 21.01743507385254, "learning_rate": 9.963636363636364e-06, "num_tokens": 25175.0, "completions/mean_length": 44.875, "completions/min_length": 30.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.875, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.7244622707366943, "rewards/meter/std": 0.4260440170764923, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9886103868484497, "rewards/repeat_soft/std": 0.017639046534895897, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.10507651418447495, "rewards/total_composite/mean": 0.6614029407501221, "rewards/total_composite/std": 0.2924328148365021, "reward": 0.6614029407501221, "reward_std": 0.2924328148365021, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21056580543518066, "sampling/sampling_logp_difference/max": 1.903547763824463, "sampling/importance_sampling_ratio/min": 0.1796707957983017, "sampling/importance_sampling_ratio/mean": 1.0298173427581787, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.938667967915535, "clip_ratio/low_mean": 0.06033376231789589, "clip_ratio/low_min": 0.06033376231789589, "clip_ratio/high_mean": 0.13869474548846483, "clip_ratio/high_max": 0.13869474548846483, "clip_ratio/region_mean": 0.19902850780636072, "reward_total_mean": 0.6614029407501221, "reward_meter_mean": 0.7244622707366943, "reward_meter_std": 0.4260440170764923, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9886103868484497, "reward_repeat_soft_std": 0.017639046534895897, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.10507651418447495, "reward_total_composite_mean": 0.6614029407501221, "reward_total_composite_std": 0.2924328148365021} {"timestamp_utc": "2026-04-12T22:15:58Z", "mode": "train", "global_step": 14, "epoch": 0.0014063284781516826, "loss": 0.1867, "grad_norm": 9.992451667785645, "learning_rate": 9.960606060606062e-06, "num_tokens": 27778.0, "completions/mean_length": 131.375, "completions/min_length": 89.0, "completions/max_length": 192.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.375, "completions/min_terminated_length": 89.0, "completions/max_terminated_length": 192.0, "rewards/meter/mean": 0.8847721815109253, "rewards/meter/std": 0.2510175108909607, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9979761838912964, "rewards/repeat_soft/std": 0.0023726883810013533, "rewards/judge_quality/mean": 0.5737500190734863, "rewards/judge_quality/std": 0.21836651861667633, "rewards/total_composite/mean": 0.8200700283050537, "rewards/total_composite/std": 0.16812320053577423, "reward": 0.8200700283050537, "reward_std": 0.16812318563461304, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2101648896932602, "sampling/sampling_logp_difference/max": 1.9604809284210205, "sampling/importance_sampling_ratio/min": 0.14079070091247559, "sampling/importance_sampling_ratio/mean": 1.0326483249664307, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4273063838481903, "clip_ratio/low_mean": 0.05659673362970352, "clip_ratio/low_min": 0.05659673362970352, "clip_ratio/high_mean": 0.15316884219646454, "clip_ratio/high_max": 0.15316884219646454, "clip_ratio/region_mean": 0.20976557582616806, "reward_total_mean": 0.8200700283050537, "reward_meter_mean": 0.8847721815109253, "reward_meter_std": 0.2510175108909607, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9979761838912964, "reward_repeat_soft_std": 0.0023726883810013533, "reward_judge_quality_mean": 0.5737500190734863, "reward_judge_quality_std": 0.21836651861667633, "reward_total_composite_mean": 0.8200700283050537, "reward_total_composite_std": 0.16812320053577423} {"timestamp_utc": "2026-04-12T22:16:04Z", "mode": "train", "global_step": 15, "epoch": 0.0015067805123053742, "loss": 0.0293, "grad_norm": 16.65377426147461, "learning_rate": 9.957575757575757e-06, "num_tokens": 29486.0, "completions/mean_length": 48.5, "completions/min_length": 25.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.5, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.4189527928829193, "rewards/meter/std": 0.3443324863910675, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9987918138504028, "rewards/repeat_soft/std": 0.00322684645652771, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.22720351815223694, "rewards/total_composite/mean": 0.5490054488182068, "rewards/total_composite/std": 0.2890552580356598, "reward": 0.5490054488182068, "reward_std": 0.2890552580356598, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19479593634605408, "sampling/sampling_logp_difference/max": 2.340357780456543, "sampling/importance_sampling_ratio/min": 0.09629317373037338, "sampling/importance_sampling_ratio/mean": 1.0266320705413818, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6842268258333206, "clip_ratio/low_mean": 0.10162054747343063, "clip_ratio/low_min": 0.10162054747343063, "clip_ratio/high_mean": 0.11235708184540272, "clip_ratio/high_max": 0.11235708184540272, "clip_ratio/region_mean": 0.21397762931883335, "reward_total_mean": 0.5490054488182068, "reward_meter_mean": 0.4189527928829193, "reward_meter_std": 0.3443324863910675, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9987918138504028, "reward_repeat_soft_std": 0.00322684645652771, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.22720351815223694, "reward_total_composite_mean": 0.5490054488182068, "reward_total_composite_std": 0.2890552580356598} {"timestamp_utc": "2026-04-12T22:16:10Z", "mode": "train", "global_step": 16, "epoch": 0.0016072325464590658, "loss": -0.0032, "grad_norm": 16.885639190673828, "learning_rate": 9.954545454545456e-06, "num_tokens": 31057.0, "completions/mean_length": 45.375, "completions/min_length": 27.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.375, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.43825337290763855, "rewards/meter/std": 0.4620993435382843, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9892818331718445, "rewards/repeat_soft/std": 0.01204143650829792, "rewards/judge_quality/mean": 0.4937500059604645, "rewards/judge_quality/std": 0.1728696972131729, "rewards/total_composite/mean": 0.5848921537399292, "rewards/total_composite/std": 0.20833450555801392, "reward": 0.5848921537399292, "reward_std": 0.20833447575569153, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.27420946955680847, "sampling/sampling_logp_difference/max": 2.7227797508239746, "sampling/importance_sampling_ratio/min": 0.06569189578294754, "sampling/importance_sampling_ratio/mean": 1.0040487051010132, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9974622428417206, "clip_ratio/low_mean": 0.07798896264284849, "clip_ratio/low_min": 0.07798896264284849, "clip_ratio/high_mean": 0.11831947416067123, "clip_ratio/high_max": 0.11831947416067123, "clip_ratio/region_mean": 0.19630843680351973, "reward_total_mean": 0.5848921537399292, "reward_meter_mean": 0.43825337290763855, "reward_meter_std": 0.4620993435382843, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9892818331718445, "reward_repeat_soft_std": 0.01204143650829792, "reward_judge_quality_mean": 0.4937500059604645, "reward_judge_quality_std": 0.1728696972131729, "reward_total_composite_mean": 0.5848921537399292, "reward_total_composite_std": 0.20833450555801392} {"timestamp_utc": "2026-04-12T22:16:15Z", "mode": "train", "global_step": 17, "epoch": 0.0017076845806127574, "loss": 0.081, "grad_norm": 18.933856964111328, "learning_rate": 9.951515151515152e-06, "num_tokens": 32706.0, "completions/mean_length": 47.125, "completions/min_length": 31.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.125, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.5159446001052856, "rewards/meter/std": 0.4337084889411926, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9935091733932495, "rewards/repeat_soft/std": 0.0063369860872626305, "rewards/judge_quality/mean": 0.5475000143051147, "rewards/judge_quality/std": 0.14320316910743713, "rewards/total_composite/mean": 0.6457759737968445, "rewards/total_composite/std": 0.2059926986694336, "reward": 0.6457759737968445, "reward_std": 0.2059926986694336, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21604323387145996, "sampling/sampling_logp_difference/max": 1.6152591705322266, "sampling/importance_sampling_ratio/min": 0.19883912801742554, "sampling/importance_sampling_ratio/mean": 1.0173397064208984, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0722051858901978, "clip_ratio/low_mean": 0.11042729578912258, "clip_ratio/low_min": 0.11042729578912258, "clip_ratio/high_mean": 0.11535557918250561, "clip_ratio/high_max": 0.11535557918250561, "clip_ratio/region_mean": 0.2257828749716282, "reward_total_mean": 0.6457759737968445, "reward_meter_mean": 0.5159446001052856, "reward_meter_std": 0.4337084889411926, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9935091733932495, "reward_repeat_soft_std": 0.0063369860872626305, "reward_judge_quality_mean": 0.5475000143051147, "reward_judge_quality_std": 0.14320316910743713, "reward_total_composite_mean": 0.6457759737968445, "reward_total_composite_std": 0.2059926986694336} {"timestamp_utc": "2026-04-12T22:16:21Z", "mode": "train", "global_step": 18, "epoch": 0.001808136614766449, "loss": -0.0019, "grad_norm": 17.924867630004883, "learning_rate": 9.948484848484849e-06, "num_tokens": 34138.0, "completions/mean_length": 30.0, "completions/min_length": 19.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.0, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.583729088306427, "rewards/meter/std": 0.350315123796463, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.4087499976158142, "rewards/judge_quality/std": 0.10507649928331375, "rewards/total_composite/mean": 0.631553053855896, "rewards/total_composite/std": 0.17265585064888, "reward": 0.631553053855896, "reward_std": 0.17265585064888, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20634080469608307, "sampling/sampling_logp_difference/max": 1.4612393379211426, "sampling/importance_sampling_ratio/min": 0.2319486439228058, "sampling/importance_sampling_ratio/mean": 1.0241836309432983, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7248151153326035, "clip_ratio/low_mean": 0.10481186397373676, "clip_ratio/low_min": 0.10481186397373676, "clip_ratio/high_mean": 0.12524540536105633, "clip_ratio/high_max": 0.12524540536105633, "clip_ratio/region_mean": 0.2300572693347931, "reward_total_mean": 0.631553053855896, "reward_meter_mean": 0.583729088306427, "reward_meter_std": 0.350315123796463, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.4087499976158142, "reward_judge_quality_std": 0.10507649928331375, "reward_total_composite_mean": 0.631553053855896, "reward_total_composite_std": 0.17265585064888} {"timestamp_utc": "2026-04-12T22:16:28Z", "mode": "train", "global_step": 19, "epoch": 0.0019085886489201406, "loss": -0.0696, "grad_norm": 14.65036392211914, "learning_rate": 9.945454545454546e-06, "num_tokens": 35798.0, "completions/mean_length": 52.5, "completions/min_length": 32.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.5, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.8749610185623169, "rewards/meter/std": 0.2477712482213974, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9977357387542725, "rewards/repeat_soft/std": 0.002847629599273205, "rewards/judge_quality/mean": 0.5550000071525574, "rewards/judge_quality/std": 0.33393755555152893, "rewards/total_composite/mean": 0.8100060224533081, "rewards/total_composite/std": 0.1606670320034027, "reward": 0.8100060224533081, "reward_std": 0.1606670320034027, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20267684757709503, "sampling/sampling_logp_difference/max": 1.836294412612915, "sampling/importance_sampling_ratio/min": 0.15940703451633453, "sampling/importance_sampling_ratio/mean": 1.0041242837905884, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5146896243095398, "clip_ratio/low_mean": 0.041744403541088104, "clip_ratio/low_min": 0.041744403541088104, "clip_ratio/high_mean": 0.1479868022724986, "clip_ratio/high_max": 0.1479868022724986, "clip_ratio/region_mean": 0.1897312058135867, "reward_total_mean": 0.8100060224533081, "reward_meter_mean": 0.8749610185623169, "reward_meter_std": 0.2477712482213974, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9977357387542725, "reward_repeat_soft_std": 0.002847629599273205, "reward_judge_quality_mean": 0.5550000071525574, "reward_judge_quality_std": 0.33393755555152893, "reward_total_composite_mean": 0.8100060224533081, "reward_total_composite_std": 0.1606670320034027} {"timestamp_utc": "2026-04-12T22:16:35Z", "mode": "train", "global_step": 20, "epoch": 0.0020090406830738324, "loss": 0.1452, "grad_norm": 12.367934226989746, "learning_rate": 9.942424242424244e-06, "num_tokens": 38169.0, "completions/mean_length": 109.375, "completions/min_length": 66.0, "completions/max_length": 165.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 109.375, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 165.0, "rewards/meter/mean": 0.7363971471786499, "rewards/meter/std": 0.274208128452301, "rewards/count_adherence/mean": 0.8333333730697632, "rewards/count_adherence/std": 0.2519763112068176, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9973340034484863, "rewards/repeat_soft/std": 0.002601567655801773, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.6832370758056641, "rewards/total_composite/std": 0.1437705010175705, "reward": 0.6832370758056641, "reward_std": 0.1437705010175705, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20733380317687988, "sampling/sampling_logp_difference/max": 1.9372377395629883, "sampling/importance_sampling_ratio/min": 0.14410145580768585, "sampling/importance_sampling_ratio/mean": 1.012389898300171, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.014356479048729, "clip_ratio/low_mean": 0.09246364235877991, "clip_ratio/low_min": 0.09246364235877991, "clip_ratio/high_mean": 0.11669936217367649, "clip_ratio/high_max": 0.11669936217367649, "clip_ratio/region_mean": 0.2091630045324564, "reward_total_mean": 0.6832370758056641, "reward_meter_mean": 0.7363971471786499, "reward_meter_std": 0.274208128452301, "reward_count_adherence_mean": 0.8333333730697632, "reward_count_adherence_std": 0.2519763112068176, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9973340034484863, "reward_repeat_soft_std": 0.002601567655801773, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.6832370758056641, "reward_total_composite_std": 0.1437705010175705} {"timestamp_utc": "2026-04-12T22:16:40Z", "mode": "train", "global_step": 21, "epoch": 0.002109492717227524, "loss": 0.0954, "grad_norm": 33.46713638305664, "learning_rate": 9.939393939393939e-06, "num_tokens": 39729.0, "completions/mean_length": 36.0, "completions/min_length": 26.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.0, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.24060703814029694, "rewards/meter/std": 0.28769800066947937, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9997790455818176, "rewards/repeat_soft/std": 0.0006249745492823422, "rewards/judge_quality/mean": 0.7987500429153442, "rewards/judge_quality/std": 0.22465451061725616, "rewards/total_composite/mean": 0.5978760719299316, "rewards/total_composite/std": 0.15713509917259216, "reward": 0.5978760719299316, "reward_std": 0.15713508427143097, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2637569010257721, "sampling/sampling_logp_difference/max": 2.1834075450897217, "sampling/importance_sampling_ratio/min": 0.11265699565410614, "sampling/importance_sampling_ratio/mean": 1.0002814531326294, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2168454974889755, "clip_ratio/low_mean": 0.18306511640548706, "clip_ratio/low_min": 0.18306511640548706, "clip_ratio/high_mean": 0.03896103985607624, "clip_ratio/high_max": 0.03896103985607624, "clip_ratio/region_mean": 0.2220261562615633, "reward_total_mean": 0.5978760719299316, "reward_meter_mean": 0.24060703814029694, "reward_meter_std": 0.28769800066947937, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9997790455818176, "reward_repeat_soft_std": 0.0006249745492823422, "reward_judge_quality_mean": 0.7987500429153442, "reward_judge_quality_std": 0.22465451061725616, "reward_total_composite_mean": 0.5978760719299316, "reward_total_composite_std": 0.15713509917259216} {"timestamp_utc": "2026-04-12T22:16:46Z", "mode": "train", "global_step": 22, "epoch": 0.0022099447513812156, "loss": 0.0519, "grad_norm": 14.068692207336426, "learning_rate": 9.936363636363638e-06, "num_tokens": 41479.0, "completions/mean_length": 59.75, "completions/min_length": 39.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.75, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9757623672485352, "rewards/meter/std": 0.020018933340907097, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9967460632324219, "rewards/repeat_soft/std": 0.005034734960645437, "rewards/judge_quality/mean": 0.6024999618530273, "rewards/judge_quality/std": 0.2521762549877167, "rewards/total_composite/mean": 0.8695176839828491, "rewards/total_composite/std": 0.07778604328632355, "reward": 0.8695176839828491, "reward_std": 0.07778605818748474, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20508892834186554, "sampling/sampling_logp_difference/max": 1.620192527770996, "sampling/importance_sampling_ratio/min": 0.19786059856414795, "sampling/importance_sampling_ratio/mean": 1.0207918882369995, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.071342244744301, "clip_ratio/low_mean": 0.10649552661925554, "clip_ratio/low_min": 0.10649552661925554, "clip_ratio/high_mean": 0.07769765146076679, "clip_ratio/high_max": 0.07769765146076679, "clip_ratio/region_mean": 0.18419317808002234, "reward_total_mean": 0.8695176839828491, "reward_meter_mean": 0.9757623672485352, "reward_meter_std": 0.020018933340907097, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9967460632324219, "reward_repeat_soft_std": 0.005034734960645437, "reward_judge_quality_mean": 0.6024999618530273, "reward_judge_quality_std": 0.2521762549877167, "reward_total_composite_mean": 0.8695176839828491, "reward_total_composite_std": 0.07778604328632355} {"timestamp_utc": "2026-04-12T22:16:52Z", "mode": "train", "global_step": 23, "epoch": 0.0023103967855349072, "loss": 0.0951, "grad_norm": 25.26443099975586, "learning_rate": 9.933333333333334e-06, "num_tokens": 43076.0, "completions/mean_length": 39.625, "completions/min_length": 35.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.625, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.41748276352882385, "rewards/meter/std": 0.4426417350769043, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9921599626541138, "rewards/repeat_soft/std": 0.009428190067410469, "rewards/judge_quality/mean": 0.5612500309944153, "rewards/judge_quality/std": 0.1968638300895691, "rewards/total_composite/mean": 0.6054582595825195, "rewards/total_composite/std": 0.19344164431095123, "reward": 0.6054582595825195, "reward_std": 0.19344164431095123, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1857699751853943, "sampling/sampling_logp_difference/max": 1.0207924842834473, "sampling/importance_sampling_ratio/min": 0.36030930280685425, "sampling/importance_sampling_ratio/mean": 1.0526957511901855, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6756559312343597, "clip_ratio/low_mean": 0.11134098190814257, "clip_ratio/low_min": 0.11134098190814257, "clip_ratio/high_mean": 0.04910453222692013, "clip_ratio/high_max": 0.04910453222692013, "clip_ratio/region_mean": 0.1604455141350627, "reward_total_mean": 0.6054582595825195, "reward_meter_mean": 0.41748276352882385, "reward_meter_std": 0.4426417350769043, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9921599626541138, "reward_repeat_soft_std": 0.009428190067410469, "reward_judge_quality_mean": 0.5612500309944153, "reward_judge_quality_std": 0.1968638300895691, "reward_total_composite_mean": 0.6054582595825195, "reward_total_composite_std": 0.19344164431095123} {"timestamp_utc": "2026-04-12T22:16:58Z", "mode": "train", "global_step": 24, "epoch": 0.002410848819688599, "loss": 0.0564, "grad_norm": 18.268474578857422, "learning_rate": 9.930303030303031e-06, "num_tokens": 44791.0, "completions/mean_length": 62.375, "completions/min_length": 42.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.375, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.634819746017456, "rewards/meter/std": 0.4580250084400177, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9998894929885864, "rewards/repeat_soft/std": 0.00031247673905454576, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.7179077863693237, "rewards/total_composite/std": 0.2593117356300354, "reward": 0.7179077863693237, "reward_std": 0.2593117356300354, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21290218830108643, "sampling/sampling_logp_difference/max": 1.9016647338867188, "sampling/importance_sampling_ratio/min": 0.1493198424577713, "sampling/importance_sampling_ratio/mean": 1.0068492889404297, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.353201448917389, "clip_ratio/low_mean": 0.06295787636190653, "clip_ratio/low_min": 0.06295787636190653, "clip_ratio/high_mean": 0.107818647287786, "clip_ratio/high_max": 0.107818647287786, "clip_ratio/region_mean": 0.17077652364969254, "reward_total_mean": 0.7179077863693237, "reward_meter_mean": 0.634819746017456, "reward_meter_std": 0.4580250084400177, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9998894929885864, "reward_repeat_soft_std": 0.00031247673905454576, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.7179077863693237, "reward_total_composite_std": 0.2593117356300354} {"timestamp_utc": "2026-04-12T22:17:05Z", "mode": "train", "global_step": 25, "epoch": 0.0025113008538422904, "loss": 0.0743, "grad_norm": 8.258381843566895, "learning_rate": 9.927272727272728e-06, "num_tokens": 47647.0, "completions/mean_length": 148.0, "completions/min_length": 129.0, "completions/max_length": 173.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 148.0, "completions/min_terminated_length": 129.0, "completions/max_terminated_length": 173.0, "rewards/meter/mean": 0.5918176174163818, "rewards/meter/std": 0.33095410466194153, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9976328611373901, "rewards/repeat_soft/std": 0.0029239931609481573, "rewards/judge_quality/mean": 0.7024999856948853, "rewards/judge_quality/std": 0.24294327199459076, "rewards/total_composite/mean": 0.7193312048912048, "rewards/total_composite/std": 0.1843627542257309, "reward": 0.7193312048912048, "reward_std": 0.1843627244234085, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16058434545993805, "sampling/sampling_logp_difference/max": 2.274415969848633, "sampling/importance_sampling_ratio/min": 0.10285695642232895, "sampling/importance_sampling_ratio/mean": 1.0136536359786987, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0220863968133926, "clip_ratio/low_mean": 0.06519535928964615, "clip_ratio/low_min": 0.06519535928964615, "clip_ratio/high_mean": 0.06505231745541096, "clip_ratio/high_max": 0.06505231745541096, "clip_ratio/region_mean": 0.1302476767450571, "reward_total_mean": 0.7193312048912048, "reward_meter_mean": 0.5918176174163818, "reward_meter_std": 0.33095410466194153, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9976328611373901, "reward_repeat_soft_std": 0.0029239931609481573, "reward_judge_quality_mean": 0.7024999856948853, "reward_judge_quality_std": 0.24294327199459076, "reward_total_composite_mean": 0.7193312048912048, "reward_total_composite_std": 0.1843627542257309} {"timestamp_utc": "2026-04-12T22:17:12Z", "mode": "train", "global_step": 26, "epoch": 0.002611752887995982, "loss": -0.1294, "grad_norm": 24.321762084960938, "learning_rate": 9.924242424242425e-06, "num_tokens": 49143.0, "completions/mean_length": 30.0, "completions/min_length": 18.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.0, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9500245451927185, "rewards/meter/std": 0.13721725344657898, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7997610569000244, "rewards/total_composite/std": 0.06174775958061218, "reward": 0.7997610569000244, "reward_std": 0.06174774840474129, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0748242437839508, "sampling/sampling_logp_difference/max": 1.1188440322875977, "sampling/importance_sampling_ratio/min": 0.32665717601776123, "sampling/importance_sampling_ratio/mean": 0.9947167038917542, "sampling/importance_sampling_ratio/max": 1.8977136611938477, "entropy": 0.4834756925702095, "clip_ratio/low_mean": 0.02777777798473835, "clip_ratio/low_min": 0.02777777798473835, "clip_ratio/high_mean": 0.030977724120020866, "clip_ratio/high_max": 0.030977724120020866, "clip_ratio/region_mean": 0.058755502104759216, "reward_total_mean": 0.7997610569000244, "reward_meter_mean": 0.9500245451927185, "reward_meter_std": 0.13721725344657898, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7997610569000244, "reward_total_composite_std": 0.06174775958061218} {"timestamp_utc": "2026-04-12T22:17:19Z", "mode": "train", "global_step": 27, "epoch": 0.0027122049221496736, "loss": -0.0948, "grad_norm": 7.565678596496582, "learning_rate": 9.921212121212121e-06, "num_tokens": 51958.0, "completions/mean_length": 154.875, "completions/min_length": 95.0, "completions/max_length": 189.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 154.875, "completions/min_terminated_length": 95.0, "completions/max_terminated_length": 189.0, "rewards/meter/mean": 0.611936628818512, "rewards/meter/std": 0.3855309784412384, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9979890584945679, "rewards/repeat_soft/std": 0.0016030868282541633, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5971717834472656, "rewards/total_composite/std": 0.28371596336364746, "reward": 0.5971717834472656, "reward_std": 0.28371599316596985, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21463504433631897, "sampling/sampling_logp_difference/max": 2.0930118560791016, "sampling/importance_sampling_ratio/min": 0.12331517040729523, "sampling/importance_sampling_ratio/mean": 1.0371588468551636, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4359234273433685, "clip_ratio/low_mean": 0.07277752831578255, "clip_ratio/low_min": 0.07277752831578255, "clip_ratio/high_mean": 0.1326075941324234, "clip_ratio/high_max": 0.1326075941324234, "clip_ratio/region_mean": 0.20538512244820595, "reward_total_mean": 0.5971717834472656, "reward_meter_mean": 0.611936628818512, "reward_meter_std": 0.3855309784412384, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9979890584945679, "reward_repeat_soft_std": 0.0016030868282541633, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5971717834472656, "reward_total_composite_std": 0.28371596336364746} {"timestamp_utc": "2026-04-12T22:17:24Z", "mode": "train", "global_step": 28, "epoch": 0.0028126569563033652, "loss": 0.0946, "grad_norm": 23.491594314575195, "learning_rate": 9.918181818181818e-06, "num_tokens": 53496.0, "completions/mean_length": 30.25, "completions/min_length": 20.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.25, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.5963934063911438, "rewards/meter/std": 0.4609495997428894, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9303162097930908, "rewards/repeat_soft/std": 0.06846698373556137, "rewards/judge_quality/mean": 0.4424999952316284, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.644158661365509, "rewards/total_composite/std": 0.20341292023658752, "reward": 0.644158661365509, "reward_std": 0.20341289043426514, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22669285535812378, "sampling/sampling_logp_difference/max": 1.899033546447754, "sampling/importance_sampling_ratio/min": 0.14971324801445007, "sampling/importance_sampling_ratio/mean": 1.014909267425537, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6166850328445435, "clip_ratio/low_mean": 0.0859464481472969, "clip_ratio/low_min": 0.0859464481472969, "clip_ratio/high_mean": 0.08879269333556294, "clip_ratio/high_max": 0.08879269333556294, "clip_ratio/region_mean": 0.17473914148285985, "reward_total_mean": 0.644158661365509, "reward_meter_mean": 0.5963934063911438, "reward_meter_std": 0.4609495997428894, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9303162097930908, "reward_repeat_soft_std": 0.06846698373556137, "reward_judge_quality_mean": 0.4424999952316284, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.644158661365509, "reward_total_composite_std": 0.20341292023658752} {"timestamp_utc": "2026-04-12T22:17:32Z", "mode": "train", "global_step": 29, "epoch": 0.002913108990457057, "loss": 0.0661, "grad_norm": 7.243879795074463, "learning_rate": 9.915151515151515e-06, "num_tokens": 56603.0, "completions/mean_length": 188.375, "completions/min_length": 179.0, "completions/max_length": 204.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 188.375, "completions/min_terminated_length": 179.0, "completions/max_terminated_length": 204.0, "rewards/meter/mean": 0.6960962414741516, "rewards/meter/std": 0.3066192865371704, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9989635348320007, "rewards/repeat_soft/std": 0.000937586824875325, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.6790146827697754, "rewards/total_composite/std": 0.14062994718551636, "reward": 0.6790146827697754, "reward_std": 0.14062994718551636, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19934551417827606, "sampling/sampling_logp_difference/max": 1.6785790920257568, "sampling/importance_sampling_ratio/min": 0.1866389811038971, "sampling/importance_sampling_ratio/mean": 1.0356954336166382, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.226356267929077, "clip_ratio/low_mean": 0.06664236634969711, "clip_ratio/low_min": 0.06664236634969711, "clip_ratio/high_mean": 0.12853810749948025, "clip_ratio/high_max": 0.12853810749948025, "clip_ratio/region_mean": 0.19518047384917736, "reward_total_mean": 0.6790146827697754, "reward_meter_mean": 0.6960962414741516, "reward_meter_std": 0.3066192865371704, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9989635348320007, "reward_repeat_soft_std": 0.000937586824875325, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.6790146827697754, "reward_total_composite_std": 0.14062994718551636} {"timestamp_utc": "2026-04-12T22:17:38Z", "mode": "train", "global_step": 30, "epoch": 0.0030135610246107484, "loss": 0.0748, "grad_norm": 11.622523307800293, "learning_rate": 9.912121212121213e-06, "num_tokens": 58613.0, "completions/mean_length": 67.25, "completions/min_length": 48.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.25, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.6762853264808655, "rewards/meter/std": 0.3140008747577667, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9974287152290344, "rewards/repeat_soft/std": 0.003936082124710083, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.6988212466239929, "rewards/total_composite/std": 0.10489797592163086, "reward": 0.6988212466239929, "reward_std": 0.10489796847105026, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20940160751342773, "sampling/sampling_logp_difference/max": 2.6506054401397705, "sampling/importance_sampling_ratio/min": 0.07060845196247101, "sampling/importance_sampling_ratio/mean": 1.0177046060562134, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0744355469942093, "clip_ratio/low_mean": 0.08076691627502441, "clip_ratio/low_min": 0.08076691627502441, "clip_ratio/high_mean": 0.1163441464304924, "clip_ratio/high_max": 0.1163441464304924, "clip_ratio/region_mean": 0.19711106270551682, "reward_total_mean": 0.6988212466239929, "reward_meter_mean": 0.6762853264808655, "reward_meter_std": 0.3140008747577667, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9974287152290344, "reward_repeat_soft_std": 0.003936082124710083, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.6988212466239929, "reward_total_composite_std": 0.10489797592163086} {"timestamp_utc": "2026-04-12T22:17:45Z", "mode": "train", "global_step": 31, "epoch": 0.00311401305876444, "loss": 0.0371, "grad_norm": 21.77640151977539, "learning_rate": 9.90909090909091e-06, "num_tokens": 60623.0, "completions/mean_length": 79.25, "completions/min_length": 61.0, "completions/max_length": 95.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.25, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.3762175440788269, "rewards/meter/std": 0.3404117226600647, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.995638370513916, "rewards/repeat_soft/std": 0.0033319119829684496, "rewards/judge_quality/mean": 0.6112500429153442, "rewards/judge_quality/std": 0.25587037205696106, "rewards/total_composite/mean": 0.4848466217517853, "rewards/total_composite/std": 0.30984121561050415, "reward": 0.4848466217517853, "reward_std": 0.30984121561050415, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2795345187187195, "sampling/sampling_logp_difference/max": 3.1914567947387695, "sampling/importance_sampling_ratio/min": 0.041111934930086136, "sampling/importance_sampling_ratio/mean": 1.0121794939041138, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1605252847075462, "clip_ratio/low_mean": 0.05587913282215595, "clip_ratio/low_min": 0.05587913282215595, "clip_ratio/high_mean": 0.14557625632733107, "clip_ratio/high_max": 0.14557625632733107, "clip_ratio/region_mean": 0.20145538914948702, "reward_total_mean": 0.4848466217517853, "reward_meter_mean": 0.3762175440788269, "reward_meter_std": 0.3404117226600647, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.995638370513916, "reward_repeat_soft_std": 0.0033319119829684496, "reward_judge_quality_mean": 0.6112500429153442, "reward_judge_quality_std": 0.25587037205696106, "reward_total_composite_mean": 0.4848466217517853, "reward_total_composite_std": 0.30984121561050415} {"timestamp_utc": "2026-04-12T22:17:51Z", "mode": "train", "global_step": 32, "epoch": 0.0032144650929181316, "loss": 0.0928, "grad_norm": 28.076183319091797, "learning_rate": 9.906060606060607e-06, "num_tokens": 62100.0, "completions/mean_length": 29.625, "completions/min_length": 22.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.625, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.48696160316467285, "rewards/meter/std": 0.4289628863334656, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9554293155670166, "rewards/repeat_soft/std": 0.013228065334260464, "rewards/judge_quality/mean": 0.9200000166893005, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.740675687789917, "rewards/total_composite/std": 0.19279979169368744, "reward": 0.740675687789917, "reward_std": 0.19279977679252625, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24338388442993164, "sampling/sampling_logp_difference/max": 2.743772506713867, "sampling/importance_sampling_ratio/min": 0.06432721018791199, "sampling/importance_sampling_ratio/mean": 1.0007094144821167, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2935088127851486, "clip_ratio/low_mean": 0.11610591504722834, "clip_ratio/low_min": 0.11610591504722834, "clip_ratio/high_mean": 0.10445739328861237, "clip_ratio/high_max": 0.10445739328861237, "clip_ratio/region_mean": 0.2205633083358407, "reward_total_mean": 0.740675687789917, "reward_meter_mean": 0.48696160316467285, "reward_meter_std": 0.4289628863334656, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9554293155670166, "reward_repeat_soft_std": 0.013228065334260464, "reward_judge_quality_mean": 0.9200000166893005, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.740675687789917, "reward_total_composite_std": 0.19279979169368744} {"timestamp_utc": "2026-04-12T22:17:57Z", "mode": "train", "global_step": 33, "epoch": 0.0033149171270718232, "loss": 0.0988, "grad_norm": 13.971663475036621, "learning_rate": 9.903030303030305e-06, "num_tokens": 63936.0, "completions/mean_length": 65.5, "completions/min_length": 36.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.5, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.36027342081069946, "rewards/meter/std": 0.43213415145874023, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9996891021728516, "rewards/repeat_soft/std": 0.0006069060764275491, "rewards/judge_quality/mean": 0.5237500071525574, "rewards/judge_quality/std": 0.19078317284584045, "rewards/total_composite/mean": 0.5131999254226685, "rewards/total_composite/std": 0.2780666947364807, "reward": 0.5131999254226685, "reward_std": 0.2780666649341583, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19844399392604828, "sampling/sampling_logp_difference/max": 2.652165412902832, "sampling/importance_sampling_ratio/min": 0.07049839198589325, "sampling/importance_sampling_ratio/mean": 1.0268425941467285, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8619239330291748, "clip_ratio/low_mean": 0.07098640222102404, "clip_ratio/low_min": 0.07098640222102404, "clip_ratio/high_mean": 0.10280672460794449, "clip_ratio/high_max": 0.10280672460794449, "clip_ratio/region_mean": 0.17379312682896852, "reward_total_mean": 0.5131999254226685, "reward_meter_mean": 0.36027342081069946, "reward_meter_std": 0.43213415145874023, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9996891021728516, "reward_repeat_soft_std": 0.0006069060764275491, "reward_judge_quality_mean": 0.5237500071525574, "reward_judge_quality_std": 0.19078317284584045, "reward_total_composite_mean": 0.5131999254226685, "reward_total_composite_std": 0.2780666947364807} {"timestamp_utc": "2026-04-12T22:18:03Z", "mode": "train", "global_step": 34, "epoch": 0.003415369161225515, "loss": 0.1421, "grad_norm": 17.26323699951172, "learning_rate": 9.9e-06, "num_tokens": 65663.0, "completions/mean_length": 57.875, "completions/min_length": 38.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.875, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.6663380861282349, "rewards/meter/std": 0.3654095232486725, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9954054355621338, "rewards/repeat_soft/std": 0.006265631411224604, "rewards/judge_quality/mean": 0.5637500286102295, "rewards/judge_quality/std": 0.22012579441070557, "rewards/total_composite/mean": 0.7185176610946655, "rewards/total_composite/std": 0.2052834928035736, "reward": 0.7185176610946655, "reward_std": 0.2052834928035736, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2225874960422516, "sampling/sampling_logp_difference/max": 1.3435239791870117, "sampling/importance_sampling_ratio/min": 0.2609245479106903, "sampling/importance_sampling_ratio/mean": 1.0131149291992188, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8193220719695091, "clip_ratio/low_mean": 0.09162838384509087, "clip_ratio/low_min": 0.09162838384509087, "clip_ratio/high_mean": 0.12510230764746666, "clip_ratio/high_max": 0.12510230764746666, "clip_ratio/region_mean": 0.21673069149255753, "reward_total_mean": 0.7185176610946655, "reward_meter_mean": 0.6663380861282349, "reward_meter_std": 0.3654095232486725, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9954054355621338, "reward_repeat_soft_std": 0.006265631411224604, "reward_judge_quality_mean": 0.5637500286102295, "reward_judge_quality_std": 0.22012579441070557, "reward_total_composite_mean": 0.7185176610946655, "reward_total_composite_std": 0.2052834928035736} {"timestamp_utc": "2026-04-12T22:18:09Z", "mode": "train", "global_step": 35, "epoch": 0.0035158211953792064, "loss": 0.1592, "grad_norm": 31.29047203063965, "learning_rate": 9.896969696969699e-06, "num_tokens": 67253.0, "completions/mean_length": 33.75, "completions/min_length": 28.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.75, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.5175750255584717, "rewards/meter/std": 0.333164244890213, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9875191450119019, "rewards/repeat_soft/std": 0.016329128295183182, "rewards/judge_quality/mean": 0.5974999666213989, "rewards/judge_quality/std": 0.211643248796463, "rewards/total_composite/mean": 0.6191242337226868, "rewards/total_composite/std": 0.29068228602409363, "reward": 0.6191242337226868, "reward_std": 0.29068228602409363, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20600245893001556, "sampling/sampling_logp_difference/max": 2.2265100479125977, "sampling/importance_sampling_ratio/min": 0.1079043596982956, "sampling/importance_sampling_ratio/mean": 1.0101490020751953, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9016148820519447, "clip_ratio/low_mean": 0.06943389028310776, "clip_ratio/low_min": 0.06943389028310776, "clip_ratio/high_mean": 0.11182209756225348, "clip_ratio/high_max": 0.11182209756225348, "clip_ratio/region_mean": 0.18125598784536123, "reward_total_mean": 0.6191242337226868, "reward_meter_mean": 0.5175750255584717, "reward_meter_std": 0.333164244890213, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9875191450119019, "reward_repeat_soft_std": 0.016329128295183182, "reward_judge_quality_mean": 0.5974999666213989, "reward_judge_quality_std": 0.211643248796463, "reward_total_composite_mean": 0.6191242337226868, "reward_total_composite_std": 0.29068228602409363} {"timestamp_utc": "2026-04-12T22:18:17Z", "mode": "train", "global_step": 36, "epoch": 0.003616273229532898, "loss": -0.0159, "grad_norm": 13.74673080444336, "learning_rate": 9.893939393939395e-06, "num_tokens": 68856.0, "completions/mean_length": 51.375, "completions/min_length": 38.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.375, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.3229061961174011, "rewards/meter/std": 0.35688337683677673, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9967907667160034, "rewards/repeat_soft/std": 0.003882752498611808, "rewards/judge_quality/mean": 0.6887500286102295, "rewards/judge_quality/std": 0.21931307017803192, "rewards/total_composite/mean": 0.601611852645874, "rewards/total_composite/std": 0.19675835967063904, "reward": 0.601611852645874, "reward_std": 0.19675834476947784, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1898266226053238, "sampling/sampling_logp_difference/max": 1.7012615203857422, "sampling/importance_sampling_ratio/min": 0.1824532151222229, "sampling/importance_sampling_ratio/mean": 1.0139268636703491, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4139694049954414, "clip_ratio/low_mean": 0.10160249006003141, "clip_ratio/low_min": 0.10160249006003141, "clip_ratio/high_mean": 0.03077651560306549, "clip_ratio/high_max": 0.03077651560306549, "clip_ratio/region_mean": 0.1323790056630969, "reward_total_mean": 0.601611852645874, "reward_meter_mean": 0.3229061961174011, "reward_meter_std": 0.35688337683677673, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9967907667160034, "reward_repeat_soft_std": 0.003882752498611808, "reward_judge_quality_mean": 0.6887500286102295, "reward_judge_quality_std": 0.21931307017803192, "reward_total_composite_mean": 0.601611852645874, "reward_total_composite_std": 0.19675835967063904} {"timestamp_utc": "2026-04-12T22:18:23Z", "mode": "train", "global_step": 37, "epoch": 0.0037167252636865896, "loss": -0.0113, "grad_norm": 13.515311241149902, "learning_rate": 9.890909090909092e-06, "num_tokens": 70697.0, "completions/mean_length": 59.125, "completions/min_length": 47.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.125, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.765489935874939, "rewards/meter/std": 0.34022775292396545, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9889501929283142, "rewards/repeat_soft/std": 0.010796717368066311, "rewards/judge_quality/mean": 0.5062500238418579, "rewards/judge_quality/std": 0.21685661375522614, "rewards/total_composite/mean": 0.7452405691146851, "rewards/total_composite/std": 0.13115821778774261, "reward": 0.7452405691146851, "reward_std": 0.13115820288658142, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20520399510860443, "sampling/sampling_logp_difference/max": 1.5714797973632812, "sampling/importance_sampling_ratio/min": 0.20773755013942719, "sampling/importance_sampling_ratio/mean": 1.0194767713546753, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.044485777616501, "clip_ratio/low_mean": 0.09386534243822098, "clip_ratio/low_min": 0.09386534243822098, "clip_ratio/high_mean": 0.11555166356265545, "clip_ratio/high_max": 0.11555166356265545, "clip_ratio/region_mean": 0.20941700600087643, "reward_total_mean": 0.7452405691146851, "reward_meter_mean": 0.765489935874939, "reward_meter_std": 0.34022775292396545, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9889501929283142, "reward_repeat_soft_std": 0.010796717368066311, "reward_judge_quality_mean": 0.5062500238418579, "reward_judge_quality_std": 0.21685661375522614, "reward_total_composite_mean": 0.7452405691146851, "reward_total_composite_std": 0.13115821778774261} {"timestamp_utc": "2026-04-12T22:18:29Z", "mode": "train", "global_step": 38, "epoch": 0.0038171772978402812, "loss": 0.0265, "grad_norm": 11.196772575378418, "learning_rate": 9.887878787878789e-06, "num_tokens": 72445.0, "completions/mean_length": 65.5, "completions/min_length": 60.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.5, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.8452023267745972, "rewards/meter/std": 0.2854308485984802, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9985214471817017, "rewards/repeat_soft/std": 0.004181958269327879, "rewards/judge_quality/mean": 0.5062500238418579, "rewards/judge_quality/std": 0.14201989769935608, "rewards/total_composite/mean": 0.7820681929588318, "rewards/total_composite/std": 0.09378696978092194, "reward": 0.7820681929588318, "reward_std": 0.09378696978092194, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16852843761444092, "sampling/sampling_logp_difference/max": 1.3398032188415527, "sampling/importance_sampling_ratio/min": 0.2618972063064575, "sampling/importance_sampling_ratio/mean": 1.028748631477356, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.285915844142437, "clip_ratio/low_mean": 0.017307693138718605, "clip_ratio/low_min": 0.017307693138718605, "clip_ratio/high_mean": 0.14909545984119177, "clip_ratio/high_max": 0.14909545984119177, "clip_ratio/region_mean": 0.16640315297991037, "reward_total_mean": 0.7820681929588318, "reward_meter_mean": 0.8452023267745972, "reward_meter_std": 0.2854308485984802, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9985214471817017, "reward_repeat_soft_std": 0.004181958269327879, "reward_judge_quality_mean": 0.5062500238418579, "reward_judge_quality_std": 0.14201989769935608, "reward_total_composite_mean": 0.7820681929588318, "reward_total_composite_std": 0.09378696978092194} {"timestamp_utc": "2026-04-12T22:18:36Z", "mode": "train", "global_step": 39, "epoch": 0.003917629331993973, "loss": 0.0498, "grad_norm": 21.736717224121094, "learning_rate": 9.884848484848486e-06, "num_tokens": 74160.0, "completions/mean_length": 56.375, "completions/min_length": 43.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.375, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.8037891387939453, "rewards/meter/std": 0.25627055764198303, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9939441680908203, "rewards/repeat_soft/std": 0.007073741871863604, "rewards/judge_quality/mean": 0.6525000333786011, "rewards/judge_quality/std": 0.1348809152841568, "rewards/total_composite/mean": 0.8068495392799377, "rewards/total_composite/std": 0.14063221216201782, "reward": 0.8068495392799377, "reward_std": 0.14063221216201782, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18867532908916473, "sampling/sampling_logp_difference/max": 2.8137412071228027, "sampling/importance_sampling_ratio/min": 0.05998017266392708, "sampling/importance_sampling_ratio/mean": 0.9923123121261597, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6779676862061024, "clip_ratio/low_mean": 0.06264539621770382, "clip_ratio/low_min": 0.06264539621770382, "clip_ratio/high_mean": 0.10982005670666695, "clip_ratio/high_max": 0.10982005670666695, "clip_ratio/region_mean": 0.17246545292437077, "reward_total_mean": 0.8068495392799377, "reward_meter_mean": 0.8037891387939453, "reward_meter_std": 0.25627055764198303, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9939441680908203, "reward_repeat_soft_std": 0.007073741871863604, "reward_judge_quality_mean": 0.6525000333786011, "reward_judge_quality_std": 0.1348809152841568, "reward_total_composite_mean": 0.8068495392799377, "reward_total_composite_std": 0.14063221216201782} {"timestamp_utc": "2026-04-12T22:18:42Z", "mode": "train", "global_step": 40, "epoch": 0.004018081366147665, "loss": 0.1016, "grad_norm": 22.00394058227539, "learning_rate": 9.881818181818182e-06, "num_tokens": 75789.0, "completions/mean_length": 40.625, "completions/min_length": 27.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.625, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.8645680546760559, "rewards/meter/std": 0.33847635984420776, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.952668309211731, "rewards/repeat_soft/std": 0.06325025111436844, "rewards/judge_quality/mean": 0.9200000166893005, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.9103224277496338, "rewards/total_composite/std": 0.15087567269802094, "reward": 0.9103224277496338, "reward_std": 0.15087567269802094, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1935538351535797, "sampling/sampling_logp_difference/max": 2.373173952102661, "sampling/importance_sampling_ratio/min": 0.09318449348211288, "sampling/importance_sampling_ratio/mean": 1.0082100629806519, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0727061294019222, "clip_ratio/low_mean": 0.02604166604578495, "clip_ratio/low_min": 0.02604166604578495, "clip_ratio/high_mean": 0.16503243800252676, "clip_ratio/high_max": 0.16503243800252676, "clip_ratio/region_mean": 0.1910741040483117, "reward_total_mean": 0.9103224277496338, "reward_meter_mean": 0.8645680546760559, "reward_meter_std": 0.33847635984420776, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.952668309211731, "reward_repeat_soft_std": 0.06325025111436844, "reward_judge_quality_mean": 0.9200000166893005, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.9103224277496338, "reward_total_composite_std": 0.15087567269802094} {"timestamp_utc": "2026-04-12T22:18:51Z", "mode": "train", "global_step": 41, "epoch": 0.004118533400301356, "loss": -0.1876, "grad_norm": 12.826306343078613, "learning_rate": 9.87878787878788e-06, "num_tokens": 77469.0, "completions/mean_length": 50.0, "completions/min_length": 24.0, "completions/max_length": 92.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.0, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 92.0, "rewards/meter/mean": 0.6660476922988892, "rewards/meter/std": 0.40160611271858215, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.997750997543335, "rewards/repeat_soft/std": 0.005148967728018761, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.12906257808208466, "rewards/total_composite/mean": 0.6337647438049316, "rewards/total_composite/std": 0.2952870726585388, "reward": 0.6337647438049316, "reward_std": 0.29528704285621643, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20901450514793396, "sampling/sampling_logp_difference/max": 1.7976255416870117, "sampling/importance_sampling_ratio/min": 0.16569185256958008, "sampling/importance_sampling_ratio/mean": 1.0211420059204102, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4476252645254135, "clip_ratio/low_mean": 0.09334935806691647, "clip_ratio/low_min": 0.09334935806691647, "clip_ratio/high_mean": 0.13066565804183483, "clip_ratio/high_max": 0.13066565804183483, "clip_ratio/region_mean": 0.2240150161087513, "reward_total_mean": 0.6337647438049316, "reward_meter_mean": 0.6660476922988892, "reward_meter_std": 0.40160611271858215, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.997750997543335, "reward_repeat_soft_std": 0.005148967728018761, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.12906257808208466, "reward_total_composite_mean": 0.6337647438049316, "reward_total_composite_std": 0.2952870726585388} {"timestamp_utc": "2026-04-12T22:18:57Z", "mode": "train", "global_step": 42, "epoch": 0.004218985434455048, "loss": 0.0467, "grad_norm": 24.63319969177246, "learning_rate": 9.875757575757576e-06, "num_tokens": 79094.0, "completions/mean_length": 54.125, "completions/min_length": 34.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.125, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.8649886250495911, "rewards/meter/std": 0.33145052194595337, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9943860769271851, "rewards/repeat_soft/std": 0.008962815627455711, "rewards/judge_quality/mean": 0.4437499940395355, "rewards/judge_quality/std": 0.2084595113992691, "rewards/total_composite/mean": 0.7718085050582886, "rewards/total_composite/std": 0.18098533153533936, "reward": 0.7718085050582886, "reward_std": 0.18098531663417816, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2050308734178543, "sampling/sampling_logp_difference/max": 1.4114322662353516, "sampling/importance_sampling_ratio/min": 0.24379386007785797, "sampling/importance_sampling_ratio/mean": 1.0256385803222656, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6599382311105728, "clip_ratio/low_mean": 0.0513223260641098, "clip_ratio/low_min": 0.0513223260641098, "clip_ratio/high_mean": 0.13139697071164846, "clip_ratio/high_max": 0.13139697071164846, "clip_ratio/region_mean": 0.18271929677575827, "reward_total_mean": 0.7718085050582886, "reward_meter_mean": 0.8649886250495911, "reward_meter_std": 0.33145052194595337, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9943860769271851, "reward_repeat_soft_std": 0.008962815627455711, "reward_judge_quality_mean": 0.4437499940395355, "reward_judge_quality_std": 0.2084595113992691, "reward_total_composite_mean": 0.7718085050582886, "reward_total_composite_std": 0.18098533153533936} {"timestamp_utc": "2026-04-12T22:19:04Z", "mode": "train", "global_step": 43, "epoch": 0.004319437468608739, "loss": 0.0083, "grad_norm": 13.690433502197266, "learning_rate": 9.872727272727274e-06, "num_tokens": 80983.0, "completions/mean_length": 61.125, "completions/min_length": 50.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.125, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.7919253706932068, "rewards/meter/std": 0.27126315236091614, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.984146773815155, "rewards/repeat_soft/std": 0.01696406863629818, "rewards/judge_quality/mean": 0.768750011920929, "rewards/judge_quality/std": 0.21695540845394135, "rewards/total_composite/mean": 0.8354060649871826, "rewards/total_composite/std": 0.10415403544902802, "reward": 0.8354060649871826, "reward_std": 0.10415403544902802, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15504302084445953, "sampling/sampling_logp_difference/max": 1.9836196899414062, "sampling/importance_sampling_ratio/min": 0.1375703662633896, "sampling/importance_sampling_ratio/mean": 1.0255910158157349, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.252104938030243, "clip_ratio/low_mean": 0.06915658712387085, "clip_ratio/low_min": 0.06915658712387085, "clip_ratio/high_mean": 0.07177211018279195, "clip_ratio/high_max": 0.07177211018279195, "clip_ratio/region_mean": 0.1409286973066628, "reward_total_mean": 0.8354060649871826, "reward_meter_mean": 0.7919253706932068, "reward_meter_std": 0.27126315236091614, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.984146773815155, "reward_repeat_soft_std": 0.01696406863629818, "reward_judge_quality_mean": 0.768750011920929, "reward_judge_quality_std": 0.21695540845394135, "reward_total_composite_mean": 0.8354060649871826, "reward_total_composite_std": 0.10415403544902802} {"timestamp_utc": "2026-04-12T22:19:11Z", "mode": "train", "global_step": 44, "epoch": 0.004419889502762431, "loss": 0.0866, "grad_norm": 8.773218154907227, "learning_rate": 9.869696969696971e-06, "num_tokens": 83437.0, "completions/mean_length": 126.75, "completions/min_length": 103.0, "completions/max_length": 143.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 126.75, "completions/min_terminated_length": 103.0, "completions/max_terminated_length": 143.0, "rewards/meter/mean": 0.3680556118488312, "rewards/meter/std": 0.2608366906642914, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9973905086517334, "rewards/repeat_soft/std": 0.0024202882777899504, "rewards/judge_quality/mean": 0.5074999928474426, "rewards/judge_quality/std": 0.1642080694437027, "rewards/total_composite/mean": 0.5676140785217285, "rewards/total_composite/std": 0.15416587889194489, "reward": 0.5676140785217285, "reward_std": 0.1541658639907837, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19027352333068848, "sampling/sampling_logp_difference/max": 1.7497572898864746, "sampling/importance_sampling_ratio/min": 0.17381611466407776, "sampling/importance_sampling_ratio/mean": 1.0400882959365845, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.894649162888527, "clip_ratio/low_mean": 0.11630215495824814, "clip_ratio/low_min": 0.11630215495824814, "clip_ratio/high_mean": 0.07096002250909805, "clip_ratio/high_max": 0.07096002250909805, "clip_ratio/region_mean": 0.1872621774673462, "reward_total_mean": 0.5676140785217285, "reward_meter_mean": 0.3680556118488312, "reward_meter_std": 0.2608366906642914, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9973905086517334, "reward_repeat_soft_std": 0.0024202882777899504, "reward_judge_quality_mean": 0.5074999928474426, "reward_judge_quality_std": 0.1642080694437027, "reward_total_composite_mean": 0.5676140785217285, "reward_total_composite_std": 0.15416587889194489} {"timestamp_utc": "2026-04-12T22:19:17Z", "mode": "train", "global_step": 45, "epoch": 0.0045203415369161224, "loss": -0.0796, "grad_norm": 12.670366287231445, "learning_rate": 9.866666666666668e-06, "num_tokens": 85060.0, "completions/mean_length": 51.875, "completions/min_length": 35.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.875, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.8333461284637451, "rewards/meter/std": 0.27241066098213196, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9904763698577881, "rewards/repeat_soft/std": 0.016579801216721535, "rewards/judge_quality/mean": 0.7325000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.8438034057617188, "rewards/total_composite/std": 0.17045538127422333, "reward": 0.8438034057617188, "reward_std": 0.17045536637306213, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16479381918907166, "sampling/sampling_logp_difference/max": 2.0195093154907227, "sampling/importance_sampling_ratio/min": 0.13272058963775635, "sampling/importance_sampling_ratio/mean": 1.0289181470870972, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5834421515464783, "clip_ratio/low_mean": 0.09027308598160744, "clip_ratio/low_min": 0.09027308598160744, "clip_ratio/high_mean": 0.07856571674346924, "clip_ratio/high_max": 0.07856571674346924, "clip_ratio/region_mean": 0.16883880272507668, "reward_total_mean": 0.8438034057617188, "reward_meter_mean": 0.8333461284637451, "reward_meter_std": 0.27241066098213196, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9904763698577881, "reward_repeat_soft_std": 0.016579801216721535, "reward_judge_quality_mean": 0.7325000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.8438034057617188, "reward_total_composite_std": 0.17045538127422333} {"timestamp_utc": "2026-04-12T22:19:23Z", "mode": "train", "global_step": 46, "epoch": 0.0046207935710698145, "loss": -0.0367, "grad_norm": 15.993819236755371, "learning_rate": 9.863636363636364e-06, "num_tokens": 86639.0, "completions/mean_length": 41.375, "completions/min_length": 34.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.375, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.6189440488815308, "rewards/meter/std": 0.4612226188182831, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9973520040512085, "rewards/repeat_soft/std": 0.00505313603207469, "rewards/judge_quality/mean": 0.5562499761581421, "rewards/judge_quality/std": 0.2249404788017273, "rewards/total_composite/mean": 0.6951350569725037, "rewards/total_composite/std": 0.24396578967571259, "reward": 0.6951350569725037, "reward_std": 0.2439657747745514, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22641699016094208, "sampling/sampling_logp_difference/max": 1.6650819778442383, "sampling/importance_sampling_ratio/min": 0.18917514383792877, "sampling/importance_sampling_ratio/mean": 1.0454026460647583, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.159736454486847, "clip_ratio/low_mean": 0.0926593616604805, "clip_ratio/low_min": 0.0926593616604805, "clip_ratio/high_mean": 0.11545929498970509, "clip_ratio/high_max": 0.11545929498970509, "clip_ratio/region_mean": 0.20811865665018559, "reward_total_mean": 0.6951350569725037, "reward_meter_mean": 0.6189440488815308, "reward_meter_std": 0.4612226188182831, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9973520040512085, "reward_repeat_soft_std": 0.00505313603207469, "reward_judge_quality_mean": 0.5562499761581421, "reward_judge_quality_std": 0.2249404788017273, "reward_total_composite_mean": 0.6951350569725037, "reward_total_composite_std": 0.24396578967571259} {"timestamp_utc": "2026-04-12T22:19:33Z", "mode": "train", "global_step": 47, "epoch": 0.004721245605223506, "loss": 0.1663, "grad_norm": 14.117936134338379, "learning_rate": 9.860606060606061e-06, "num_tokens": 88311.0, "completions/mean_length": 60.0, "completions/min_length": 33.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.0, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.5848500728607178, "rewards/meter/std": 0.310144305229187, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9949307441711426, "rewards/repeat_soft/std": 0.0036206496879458427, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.6949256062507629, "rewards/total_composite/std": 0.19530318677425385, "reward": 0.6949256062507629, "reward_std": 0.19530318677425385, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21896903216838837, "sampling/sampling_logp_difference/max": 1.9479656219482422, "sampling/importance_sampling_ratio/min": 0.1425638049840927, "sampling/importance_sampling_ratio/mean": 0.9914653301239014, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5602813884615898, "clip_ratio/low_mean": 0.0784417437389493, "clip_ratio/low_min": 0.0784417437389493, "clip_ratio/high_mean": 0.10587121546268463, "clip_ratio/high_max": 0.10587121546268463, "clip_ratio/region_mean": 0.18431295920163393, "reward_total_mean": 0.6949256062507629, "reward_meter_mean": 0.5848500728607178, "reward_meter_std": 0.310144305229187, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9949307441711426, "reward_repeat_soft_std": 0.0036206496879458427, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.6949256062507629, "reward_total_composite_std": 0.19530318677425385} {"timestamp_utc": "2026-04-12T22:19:39Z", "mode": "train", "global_step": 48, "epoch": 0.004821697639377198, "loss": 0.1596, "grad_norm": 25.422317504882812, "learning_rate": 9.857575757575758e-06, "num_tokens": 89968.0, "completions/mean_length": 38.125, "completions/min_length": 26.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.125, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.48131677508354187, "rewards/meter/std": 0.264057993888855, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9779799580574036, "rewards/repeat_soft/std": 0.028363212943077087, "rewards/judge_quality/mean": 0.6225000023841858, "rewards/judge_quality/std": 0.24656209349632263, "rewards/total_composite/mean": 0.651140570640564, "rewards/total_composite/std": 0.1632310152053833, "reward": 0.651140570640564, "reward_std": 0.1632310003042221, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24116353690624237, "sampling/sampling_logp_difference/max": 1.608734130859375, "sampling/importance_sampling_ratio/min": 0.2001408040523529, "sampling/importance_sampling_ratio/mean": 1.0072649717330933, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5285140573978424, "clip_ratio/low_mean": 0.13228978216648102, "clip_ratio/low_min": 0.13228978216648102, "clip_ratio/high_mean": 0.0881993044167757, "clip_ratio/high_max": 0.0881993044167757, "clip_ratio/region_mean": 0.22048908658325672, "reward_total_mean": 0.651140570640564, "reward_meter_mean": 0.48131677508354187, "reward_meter_std": 0.264057993888855, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9779799580574036, "reward_repeat_soft_std": 0.028363212943077087, "reward_judge_quality_mean": 0.6225000023841858, "reward_judge_quality_std": 0.24656209349632263, "reward_total_composite_mean": 0.651140570640564, "reward_total_composite_std": 0.1632310152053833} {"timestamp_utc": "2026-04-12T22:19:47Z", "mode": "train", "global_step": 49, "epoch": 0.004922149673530889, "loss": 0.0595, "grad_norm": 16.515077590942383, "learning_rate": 9.854545454545456e-06, "num_tokens": 92540.0, "completions/mean_length": 130.5, "completions/min_length": 119.0, "completions/max_length": 148.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 130.5, "completions/min_terminated_length": 119.0, "completions/max_terminated_length": 148.0, "rewards/meter/mean": 0.4894363284111023, "rewards/meter/std": 0.22659291326999664, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9929240345954895, "rewards/repeat_soft/std": 0.006053104996681213, "rewards/judge_quality/mean": 0.4987499713897705, "rewards/judge_quality/std": 0.13695022463798523, "rewards/total_composite/mean": 0.5284908413887024, "rewards/total_composite/std": 0.22537189722061157, "reward": 0.5284908413887024, "reward_std": 0.22537189722061157, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2464165985584259, "sampling/sampling_logp_difference/max": 3.279010057449341, "sampling/importance_sampling_ratio/min": 0.03766552358865738, "sampling/importance_sampling_ratio/mean": 0.9850641489028931, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1143654063344002, "clip_ratio/low_mean": 0.050971951335668564, "clip_ratio/low_min": 0.050971951335668564, "clip_ratio/high_mean": 0.16969024203717709, "clip_ratio/high_max": 0.16969024203717709, "clip_ratio/region_mean": 0.22066219337284565, "reward_total_mean": 0.5284908413887024, "reward_meter_mean": 0.4894363284111023, "reward_meter_std": 0.22659291326999664, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9929240345954895, "reward_repeat_soft_std": 0.006053104996681213, "reward_judge_quality_mean": 0.4987499713897705, "reward_judge_quality_std": 0.13695022463798523, "reward_total_composite_mean": 0.5284908413887024, "reward_total_composite_std": 0.22537189722061157} {"timestamp_utc": "2026-04-12T22:19:54Z", "mode": "train", "global_step": 50, "epoch": 0.005022601707684581, "loss": 0.1871, "grad_norm": 12.148331642150879, "learning_rate": 9.851515151515151e-06, "num_tokens": 94611.0, "completions/mean_length": 96.875, "completions/min_length": 60.0, "completions/max_length": 131.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.875, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.33955442905426025, "rewards/meter/std": 0.31331250071525574, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9961608648300171, "rewards/repeat_soft/std": 0.005006108433008194, "rewards/judge_quality/mean": 0.3712499737739563, "rewards/judge_quality/std": 0.21390503644943237, "rewards/total_composite/mean": 0.5137906074523926, "rewards/total_composite/std": 0.15746544301509857, "reward": 0.5137906074523926, "reward_std": 0.15746545791625977, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2518634796142578, "sampling/sampling_logp_difference/max": 2.144465446472168, "sampling/importance_sampling_ratio/min": 0.11713062971830368, "sampling/importance_sampling_ratio/mean": 1.0296486616134644, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0926555544137955, "clip_ratio/low_mean": 0.13645275123417377, "clip_ratio/low_min": 0.13645275123417377, "clip_ratio/high_mean": 0.07529207319021225, "clip_ratio/high_max": 0.07529207319021225, "clip_ratio/region_mean": 0.21174482442438602, "reward_total_mean": 0.5137906074523926, "reward_meter_mean": 0.33955442905426025, "reward_meter_std": 0.31331250071525574, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9961608648300171, "reward_repeat_soft_std": 0.005006108433008194, "reward_judge_quality_mean": 0.3712499737739563, "reward_judge_quality_std": 0.21390503644943237, "reward_total_composite_mean": 0.5137906074523926, "reward_total_composite_std": 0.15746544301509857} {"timestamp_utc": "2026-04-12T22:20:44Z", "mode": "eval", "global_step": 50, "epoch": 0.005022601707684581, "eval_loss": NaN, "eval_runtime": 50.3639, "eval_samples_per_second": 1.588, "eval_steps_per_second": 0.199, "eval_num_tokens": 94611.0, "eval_completions/mean_length": 92.5125, "eval_completions/min_length": 36.6, "eval_completions/max_length": 172.9, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 92.5125, "eval_completions/min_terminated_length": 36.6, "eval_completions/max_terminated_length": 172.9, "eval_rewards/meter/mean": 0.622799813747406, "eval_rewards/meter/std": 0.385142882168293, "eval_rewards/count_adherence/mean": 0.9683333277702332, "eval_rewards/count_adherence/std": 0.07128634452819824, "eval_rewards/hard_gate/mean": 0.95, "eval_rewards/hard_gate/std": 0.1414213538169861, "eval_rewards/repeat_soft/mean": 0.993061774969101, "eval_rewards/repeat_soft/std": 0.011147955048363656, "eval_rewards/judge_quality/mean": 0.4986249953508377, "eval_rewards/judge_quality/std": 0.16713942140340804, "eval_rewards/total_composite/mean": 0.6409295320510864, "eval_rewards/total_composite/std": 0.23319732695817946, "eval_reward": 0.6409295320510864, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.12240503132343292, "eval_sampling/sampling_logp_difference/max": 1.1358956813812255, "eval_sampling/importance_sampling_ratio/min": 0.32323764860630033, "eval_sampling/importance_sampling_ratio/mean": 1.031867229938507, "eval_sampling/importance_sampling_ratio/max": 1.530899715423584, "eval_entropy": 1.7908539295196533, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6409295320510864, "eval_reward_meter_mean": 0.622799813747406, "eval_reward_meter_std": 0.385142882168293, "eval_reward_count_adherence_mean": 0.9683333277702332, "eval_reward_count_adherence_std": 0.07128634452819824, "eval_reward_hard_gate_mean": 0.95, "eval_reward_hard_gate_std": 0.1414213538169861, "eval_reward_repeat_soft_mean": 0.993061774969101, "eval_reward_repeat_soft_std": 0.011147955048363656, "eval_reward_judge_quality_mean": 0.4986249953508377, "eval_reward_judge_quality_std": 0.16713942140340804, "eval_reward_total_composite_mean": 0.6409295320510864, "eval_reward_total_composite_std": 0.23319732695817946} {"timestamp_utc": "2026-04-12T22:20:53Z", "mode": "train", "global_step": 51, "epoch": 0.005123053741838272, "loss": 0.102, "grad_norm": 15.290033340454102, "learning_rate": 9.84848484848485e-06, "num_tokens": 96345.0, "completions/mean_length": 49.75, "completions/min_length": 35.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.75, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.8911745548248291, "rewards/meter/std": 0.18275241553783417, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9963257312774658, "rewards/repeat_soft/std": 0.005425546783953905, "rewards/judge_quality/mean": 0.5487500429153442, "rewards/judge_quality/std": 0.22937415540218353, "rewards/total_composite/mean": 0.8152861595153809, "rewards/total_composite/std": 0.12008143216371536, "reward": 0.8152861595153809, "reward_std": 0.12008143216371536, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2254175841808319, "sampling/sampling_logp_difference/max": 2.007718324661255, "sampling/importance_sampling_ratio/min": 0.13429473340511322, "sampling/importance_sampling_ratio/mean": 1.0337834358215332, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1199056655168533, "clip_ratio/low_mean": 0.07456336915493011, "clip_ratio/low_min": 0.07456336915493011, "clip_ratio/high_mean": 0.1384337618947029, "clip_ratio/high_max": 0.1384337618947029, "clip_ratio/region_mean": 0.21299713104963303, "reward_total_mean": 0.8152861595153809, "reward_meter_mean": 0.8911745548248291, "reward_meter_std": 0.18275241553783417, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9963257312774658, "reward_repeat_soft_std": 0.005425546783953905, "reward_judge_quality_mean": 0.5487500429153442, "reward_judge_quality_std": 0.22937415540218353, "reward_total_composite_mean": 0.8152861595153809, "reward_total_composite_std": 0.12008143216371536} {"timestamp_utc": "2026-04-12T22:21:00Z", "mode": "train", "global_step": 52, "epoch": 0.005223505775991964, "loss": 0.1486, "grad_norm": 10.75352668762207, "learning_rate": 9.845454545454546e-06, "num_tokens": 98656.0, "completions/mean_length": 114.875, "completions/min_length": 73.0, "completions/max_length": 159.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 114.875, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 159.0, "rewards/meter/mean": 0.7699568867683411, "rewards/meter/std": 0.3357446789741516, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9915257692337036, "rewards/repeat_soft/std": 0.004899078514426947, "rewards/judge_quality/mean": 0.6325000524520874, "rewards/judge_quality/std": 0.18850921094417572, "rewards/total_composite/mean": 0.7806956768035889, "rewards/total_composite/std": 0.20309355854988098, "reward": 0.7806956768035889, "reward_std": 0.20309355854988098, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18153929710388184, "sampling/sampling_logp_difference/max": 1.4434322118759155, "sampling/importance_sampling_ratio/min": 0.2361159771680832, "sampling/importance_sampling_ratio/mean": 1.042791724205017, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0450744926929474, "clip_ratio/low_mean": 0.051473021507263184, "clip_ratio/low_min": 0.051473021507263184, "clip_ratio/high_mean": 0.1433755587786436, "clip_ratio/high_max": 0.1433755587786436, "clip_ratio/region_mean": 0.1948485802859068, "reward_total_mean": 0.7806956768035889, "reward_meter_mean": 0.7699568867683411, "reward_meter_std": 0.3357446789741516, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9915257692337036, "reward_repeat_soft_std": 0.004899078514426947, "reward_judge_quality_mean": 0.6325000524520874, "reward_judge_quality_std": 0.18850921094417572, "reward_total_composite_mean": 0.7806956768035889, "reward_total_composite_std": 0.20309355854988098} {"timestamp_utc": "2026-04-12T22:21:06Z", "mode": "train", "global_step": 53, "epoch": 0.005323957810145655, "loss": -0.0155, "grad_norm": 24.79887580871582, "learning_rate": 9.842424242424243e-06, "num_tokens": 100184.0, "completions/mean_length": 38.0, "completions/min_length": 28.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.0, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.8915675282478333, "rewards/meter/std": 0.14790304005146027, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9969451427459717, "rewards/repeat_soft/std": 0.002551683457568288, "rewards/judge_quality/mean": 0.7325000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.7729640007019043, "rewards/total_composite/std": 0.33475446701049805, "reward": 0.7729640007019043, "reward_std": 0.33475446701049805, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21401749551296234, "sampling/sampling_logp_difference/max": 2.617398500442505, "sampling/importance_sampling_ratio/min": 0.07299250364303589, "sampling/importance_sampling_ratio/mean": 0.976485013961792, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6175712272524834, "clip_ratio/low_mean": 0.037664955481886864, "clip_ratio/low_min": 0.037664955481886864, "clip_ratio/high_mean": 0.09748073294758797, "clip_ratio/high_max": 0.09748073294758797, "clip_ratio/region_mean": 0.13514568842947483, "reward_total_mean": 0.7729640007019043, "reward_meter_mean": 0.8915675282478333, "reward_meter_std": 0.14790304005146027, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9969451427459717, "reward_repeat_soft_std": 0.002551683457568288, "reward_judge_quality_mean": 0.7325000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.7729640007019043, "reward_total_composite_std": 0.33475446701049805} {"timestamp_utc": "2026-04-12T22:21:15Z", "mode": "train", "global_step": 54, "epoch": 0.005424409844299347, "loss": 0.0579, "grad_norm": 10.786398887634277, "learning_rate": 9.83939393939394e-06, "num_tokens": 102660.0, "completions/mean_length": 132.5, "completions/min_length": 94.0, "completions/max_length": 197.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 132.5, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 197.0, "rewards/meter/mean": 0.5160332918167114, "rewards/meter/std": 0.3664347231388092, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9905648231506348, "rewards/repeat_soft/std": 0.006998918484896421, "rewards/judge_quality/mean": 0.8237500190734863, "rewards/judge_quality/std": 0.2722361385822296, "rewards/total_composite/mean": 0.5760136842727661, "rewards/total_composite/std": 0.37488675117492676, "reward": 0.5760136842727661, "reward_std": 0.37488675117492676, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21526861190795898, "sampling/sampling_logp_difference/max": 2.1233134269714355, "sampling/importance_sampling_ratio/min": 0.11963456869125366, "sampling/importance_sampling_ratio/mean": 1.02375066280365, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.026296839118004, "clip_ratio/low_mean": 0.06303927302360535, "clip_ratio/low_min": 0.06303927302360535, "clip_ratio/high_mean": 0.13252770714461803, "clip_ratio/high_max": 0.13252770714461803, "clip_ratio/region_mean": 0.19556698016822338, "reward_total_mean": 0.5760136842727661, "reward_meter_mean": 0.5160332918167114, "reward_meter_std": 0.3664347231388092, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9905648231506348, "reward_repeat_soft_std": 0.006998918484896421, "reward_judge_quality_mean": 0.8237500190734863, "reward_judge_quality_std": 0.2722361385822296, "reward_total_composite_mean": 0.5760136842727661, "reward_total_composite_std": 0.37488675117492676} {"timestamp_utc": "2026-04-12T22:21:25Z", "mode": "train", "global_step": 55, "epoch": 0.0055248618784530384, "loss": 0.09, "grad_norm": 20.380891799926758, "learning_rate": 9.836363636363637e-06, "num_tokens": 104225.0, "completions/mean_length": 28.625, "completions/min_length": 24.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.625, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.6425614356994629, "rewards/meter/std": 0.46263325214385986, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9575977325439453, "rewards/repeat_soft/std": 0.013865554705262184, "rewards/judge_quality/mean": 0.6775000095367432, "rewards/judge_quality/std": 0.25949129462242126, "rewards/total_composite/mean": 0.7381623983383179, "rewards/total_composite/std": 0.23571598529815674, "reward": 0.7381623983383179, "reward_std": 0.23571597039699554, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22299328446388245, "sampling/sampling_logp_difference/max": 1.5237703323364258, "sampling/importance_sampling_ratio/min": 0.21788883209228516, "sampling/importance_sampling_ratio/mean": 1.0329948663711548, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6725586950778961, "clip_ratio/low_mean": 0.0873684212565422, "clip_ratio/low_min": 0.0873684212565422, "clip_ratio/high_mean": 0.12113425973802805, "clip_ratio/high_max": 0.12113425973802805, "clip_ratio/region_mean": 0.20850268099457026, "reward_total_mean": 0.7381623983383179, "reward_meter_mean": 0.6425614356994629, "reward_meter_std": 0.46263325214385986, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9575977325439453, "reward_repeat_soft_std": 0.013865554705262184, "reward_judge_quality_mean": 0.6775000095367432, "reward_judge_quality_std": 0.25949129462242126, "reward_total_composite_mean": 0.7381623983383179, "reward_total_composite_std": 0.23571598529815674} {"timestamp_utc": "2026-04-12T22:21:37Z", "mode": "train", "global_step": 56, "epoch": 0.0056253139126067305, "loss": 0.044, "grad_norm": 17.16199493408203, "learning_rate": 9.833333333333333e-06, "num_tokens": 105784.0, "completions/mean_length": 36.875, "completions/min_length": 27.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.875, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.6049321889877319, "rewards/meter/std": 0.4337708652019501, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.6489694714546204, "rewards/total_composite/std": 0.19477401673793793, "reward": 0.6489694714546204, "reward_std": 0.19477401673793793, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18272539973258972, "sampling/sampling_logp_difference/max": 1.233874797821045, "sampling/importance_sampling_ratio/min": 0.2911621928215027, "sampling/importance_sampling_ratio/mean": 1.0084818601608276, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6532663255929947, "clip_ratio/low_mean": 0.10378074645996094, "clip_ratio/low_min": 0.10378074645996094, "clip_ratio/high_mean": 0.11709216423332691, "clip_ratio/high_max": 0.11709216423332691, "clip_ratio/region_mean": 0.22087291069328785, "reward_total_mean": 0.6489694714546204, "reward_meter_mean": 0.6049321889877319, "reward_meter_std": 0.4337708652019501, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.6489694714546204, "reward_total_composite_std": 0.19477401673793793} {"timestamp_utc": "2026-04-12T22:21:47Z", "mode": "train", "global_step": 57, "epoch": 0.005725765946760422, "loss": 0.1098, "grad_norm": 9.582910537719727, "learning_rate": 9.830303030303032e-06, "num_tokens": 108140.0, "completions/mean_length": 124.5, "completions/min_length": 98.0, "completions/max_length": 140.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 124.5, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 140.0, "rewards/meter/mean": 0.6479657888412476, "rewards/meter/std": 0.37889355421066284, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9963884353637695, "rewards/repeat_soft/std": 0.003934774082154036, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.6608484387397766, "rewards/total_composite/std": 0.17602038383483887, "reward": 0.6608484387397766, "reward_std": 0.17602038383483887, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19246143102645874, "sampling/sampling_logp_difference/max": 2.835052490234375, "sampling/importance_sampling_ratio/min": 0.05871544033288956, "sampling/importance_sampling_ratio/mean": 1.0382243394851685, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9649572968482971, "clip_ratio/low_mean": 0.08068174216896296, "clip_ratio/low_min": 0.08068174216896296, "clip_ratio/high_mean": 0.1007919292896986, "clip_ratio/high_max": 0.1007919292896986, "clip_ratio/region_mean": 0.18147367145866156, "reward_total_mean": 0.6608484387397766, "reward_meter_mean": 0.6479657888412476, "reward_meter_std": 0.37889355421066284, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9963884353637695, "reward_repeat_soft_std": 0.003934774082154036, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.6608484387397766, "reward_total_composite_std": 0.17602038383483887} {"timestamp_utc": "2026-04-12T22:21:58Z", "mode": "train", "global_step": 58, "epoch": 0.005826217980914114, "loss": 0.0677, "grad_norm": 10.142060279846191, "learning_rate": 9.827272727272729e-06, "num_tokens": 110448.0, "completions/mean_length": 105.5, "completions/min_length": 94.0, "completions/max_length": 118.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 105.5, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 118.0, "rewards/meter/mean": 0.6721632480621338, "rewards/meter/std": 0.34179237484931946, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9995437860488892, "rewards/repeat_soft/std": 0.0007127728313207626, "rewards/judge_quality/mean": 0.5199999809265137, "rewards/judge_quality/std": 0.19272483885288239, "rewards/total_composite/mean": 0.6385196447372437, "rewards/total_composite/std": 0.3149482011795044, "reward": 0.6385196447372437, "reward_std": 0.3149482011795044, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21328410506248474, "sampling/sampling_logp_difference/max": 1.8783636093139648, "sampling/importance_sampling_ratio/min": 0.1528400182723999, "sampling/importance_sampling_ratio/mean": 1.0194169282913208, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9980643540620804, "clip_ratio/low_mean": 0.09210692159831524, "clip_ratio/low_min": 0.09210692159831524, "clip_ratio/high_mean": 0.09094057232141495, "clip_ratio/high_max": 0.09094057232141495, "clip_ratio/region_mean": 0.1830474939197302, "reward_total_mean": 0.6385196447372437, "reward_meter_mean": 0.6721632480621338, "reward_meter_std": 0.34179237484931946, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9995437860488892, "reward_repeat_soft_std": 0.0007127728313207626, "reward_judge_quality_mean": 0.5199999809265137, "reward_judge_quality_std": 0.19272483885288239, "reward_total_composite_mean": 0.6385196447372437, "reward_total_composite_std": 0.3149482011795044} {"timestamp_utc": "2026-04-12T22:22:06Z", "mode": "train", "global_step": 59, "epoch": 0.005926670015067805, "loss": -0.021, "grad_norm": 14.809919357299805, "learning_rate": 9.824242424242425e-06, "num_tokens": 112380.0, "completions/mean_length": 62.5, "completions/min_length": 54.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.5, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.6479412317276001, "rewards/meter/std": 0.38575154542922974, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9950402975082397, "rewards/repeat_soft/std": 0.008863254450261593, "rewards/judge_quality/mean": 0.5062500238418579, "rewards/judge_quality/std": 0.2670440077781677, "rewards/total_composite/mean": 0.6347871422767639, "rewards/total_composite/std": 0.33347341418266296, "reward": 0.6347871422767639, "reward_std": 0.3334733843803406, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2120247334241867, "sampling/sampling_logp_difference/max": 1.6805639266967773, "sampling/importance_sampling_ratio/min": 0.1862688958644867, "sampling/importance_sampling_ratio/mean": 1.0009524822235107, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7054464370012283, "clip_ratio/low_mean": 0.05873700138181448, "clip_ratio/low_min": 0.05873700138181448, "clip_ratio/high_mean": 0.13117719814181328, "clip_ratio/high_max": 0.13117719814181328, "clip_ratio/region_mean": 0.18991419952362776, "reward_total_mean": 0.6347871422767639, "reward_meter_mean": 0.6479412317276001, "reward_meter_std": 0.38575154542922974, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9950402975082397, "reward_repeat_soft_std": 0.008863254450261593, "reward_judge_quality_mean": 0.5062500238418579, "reward_judge_quality_std": 0.2670440077781677, "reward_total_composite_mean": 0.6347871422767639, "reward_total_composite_std": 0.33347341418266296} {"timestamp_utc": "2026-04-12T22:22:14Z", "mode": "train", "global_step": 60, "epoch": 0.006027122049221497, "loss": 0.1206, "grad_norm": 14.03058910369873, "learning_rate": 9.821212121212122e-06, "num_tokens": 114047.0, "completions/mean_length": 63.375, "completions/min_length": 45.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.375, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.682119607925415, "rewards/meter/std": 0.40137389302253723, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9977132081985474, "rewards/repeat_soft/std": 0.00379719166085124, "rewards/judge_quality/mean": 0.7699999809265137, "rewards/judge_quality/std": 0.22677870094776154, "rewards/total_composite/mean": 0.7877250909805298, "rewards/total_composite/std": 0.19343271851539612, "reward": 0.7877250909805298, "reward_std": 0.19343270361423492, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20363926887512207, "sampling/sampling_logp_difference/max": 1.5578479766845703, "sampling/importance_sampling_ratio/min": 0.21058876812458038, "sampling/importance_sampling_ratio/mean": 1.0375148057937622, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7059862613677979, "clip_ratio/low_mean": 0.0698032584041357, "clip_ratio/low_min": 0.0698032584041357, "clip_ratio/high_mean": 0.09936619736254215, "clip_ratio/high_max": 0.09936619736254215, "clip_ratio/region_mean": 0.16916945576667786, "reward_total_mean": 0.7877250909805298, "reward_meter_mean": 0.682119607925415, "reward_meter_std": 0.40137389302253723, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9977132081985474, "reward_repeat_soft_std": 0.00379719166085124, "reward_judge_quality_mean": 0.7699999809265137, "reward_judge_quality_std": 0.22677870094776154, "reward_total_composite_mean": 0.7877250909805298, "reward_total_composite_std": 0.19343271851539612} {"timestamp_utc": "2026-04-12T22:22:21Z", "mode": "train", "global_step": 61, "epoch": 0.006127574083375188, "loss": -0.0091, "grad_norm": 13.446069717407227, "learning_rate": 9.81818181818182e-06, "num_tokens": 115837.0, "completions/mean_length": 49.75, "completions/min_length": 32.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.75, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.4147126078605652, "rewards/meter/std": 0.41163748502731323, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9938614368438721, "rewards/repeat_soft/std": 0.010955816134810448, "rewards/judge_quality/mean": 0.5012500286102295, "rewards/judge_quality/std": 0.16974246501922607, "rewards/total_composite/mean": 0.5863817930221558, "rewards/total_composite/std": 0.2178468108177185, "reward": 0.5863817930221558, "reward_std": 0.2178467959165573, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19411690533161163, "sampling/sampling_logp_difference/max": 1.7225580215454102, "sampling/importance_sampling_ratio/min": 0.17860867083072662, "sampling/importance_sampling_ratio/mean": 1.0362242460250854, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.983920469880104, "clip_ratio/low_mean": 0.1204979931935668, "clip_ratio/low_min": 0.1204979931935668, "clip_ratio/high_mean": 0.0703886691480875, "clip_ratio/high_max": 0.0703886691480875, "clip_ratio/region_mean": 0.1908866623416543, "reward_total_mean": 0.5863817930221558, "reward_meter_mean": 0.4147126078605652, "reward_meter_std": 0.41163748502731323, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9938614368438721, "reward_repeat_soft_std": 0.010955816134810448, "reward_judge_quality_mean": 0.5012500286102295, "reward_judge_quality_std": 0.16974246501922607, "reward_total_composite_mean": 0.5863817930221558, "reward_total_composite_std": 0.2178468108177185} {"timestamp_utc": "2026-04-12T22:22:30Z", "mode": "train", "global_step": 62, "epoch": 0.00622802611752888, "loss": 0.0554, "grad_norm": 8.62053108215332, "learning_rate": 9.815151515151516e-06, "num_tokens": 118853.0, "completions/mean_length": 173.0, "completions/min_length": 155.0, "completions/max_length": 207.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 173.0, "completions/min_terminated_length": 155.0, "completions/max_terminated_length": 207.0, "rewards/meter/mean": 0.4597916603088379, "rewards/meter/std": 0.29323315620422363, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.1035098284482956, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.99494469165802, "rewards/repeat_soft/std": 0.005377884954214096, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5711507201194763, "rewards/total_composite/std": 0.12056976556777954, "reward": 0.5711507201194763, "reward_std": 0.12056976556777954, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20841319859027863, "sampling/sampling_logp_difference/max": 2.5847525596618652, "sampling/importance_sampling_ratio/min": 0.07541473954916, "sampling/importance_sampling_ratio/mean": 1.0249940156936646, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5373004227876663, "clip_ratio/low_mean": 0.09918828681111336, "clip_ratio/low_min": 0.09918828681111336, "clip_ratio/high_mean": 0.11019240505993366, "clip_ratio/high_max": 0.11019240505993366, "clip_ratio/region_mean": 0.20938069187104702, "reward_total_mean": 0.5711507201194763, "reward_meter_mean": 0.4597916603088379, "reward_meter_std": 0.29323315620422363, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.1035098284482956, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.99494469165802, "reward_repeat_soft_std": 0.005377884954214096, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5711507201194763, "reward_total_composite_std": 0.12056976556777954} {"timestamp_utc": "2026-04-12T22:22:37Z", "mode": "train", "global_step": 63, "epoch": 0.006328478151682571, "loss": 0.2431, "grad_norm": 20.96488380432129, "learning_rate": 9.812121212121212e-06, "num_tokens": 120739.0, "completions/mean_length": 74.75, "completions/min_length": 57.0, "completions/max_length": 117.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.75, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.45688575506210327, "rewards/meter/std": 0.35617753863334656, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9985731840133667, "rewards/repeat_soft/std": 0.0012093555415049195, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.6039559245109558, "rewards/total_composite/std": 0.18145155906677246, "reward": 0.6039559245109558, "reward_std": 0.18145155906677246, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21291305124759674, "sampling/sampling_logp_difference/max": 2.4753305912017822, "sampling/importance_sampling_ratio/min": 0.08413517475128174, "sampling/importance_sampling_ratio/mean": 1.0076979398727417, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8964705485850573, "clip_ratio/low_mean": 0.128117217682302, "clip_ratio/low_min": 0.128117217682302, "clip_ratio/high_mean": 0.04571073269471526, "clip_ratio/high_max": 0.04571073269471526, "clip_ratio/region_mean": 0.17382795037701726, "reward_total_mean": 0.6039559245109558, "reward_meter_mean": 0.45688575506210327, "reward_meter_std": 0.35617753863334656, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9985731840133667, "reward_repeat_soft_std": 0.0012093555415049195, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.6039559245109558, "reward_total_composite_std": 0.18145155906677246} {"timestamp_utc": "2026-04-12T22:22:49Z", "mode": "train", "global_step": 64, "epoch": 0.006428930185836263, "loss": 0.0561, "grad_norm": 11.986035346984863, "learning_rate": 9.809090909090911e-06, "num_tokens": 123013.0, "completions/mean_length": 102.25, "completions/min_length": 84.0, "completions/max_length": 119.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.25, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 119.0, "rewards/meter/mean": 0.2554526627063751, "rewards/meter/std": 0.23822079598903656, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9972249269485474, "rewards/repeat_soft/std": 0.0024571451358497143, "rewards/judge_quality/mean": 0.5199999809265137, "rewards/judge_quality/std": 0.19272483885288239, "rewards/total_composite/mean": 0.43968093395233154, "rewards/total_composite/std": 0.20654380321502686, "reward": 0.43968093395233154, "reward_std": 0.20654380321502686, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23794835805892944, "sampling/sampling_logp_difference/max": 1.656651258468628, "sampling/importance_sampling_ratio/min": 0.1907767653465271, "sampling/importance_sampling_ratio/mean": 1.025471806526184, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1588610261678696, "clip_ratio/low_mean": 0.10111558996140957, "clip_ratio/low_min": 0.10111558996140957, "clip_ratio/high_mean": 0.1057747658342123, "clip_ratio/high_max": 0.1057747658342123, "clip_ratio/region_mean": 0.20689035579562187, "reward_total_mean": 0.43968093395233154, "reward_meter_mean": 0.2554526627063751, "reward_meter_std": 0.23822079598903656, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9972249269485474, "reward_repeat_soft_std": 0.0024571451358497143, "reward_judge_quality_mean": 0.5199999809265137, "reward_judge_quality_std": 0.19272483885288239, "reward_total_composite_mean": 0.43968093395233154, "reward_total_composite_std": 0.20654380321502686} {"timestamp_utc": "2026-04-12T22:22:56Z", "mode": "train", "global_step": 65, "epoch": 0.0065293822199899544, "loss": 0.0258, "grad_norm": 21.1607723236084, "learning_rate": 9.806060606060607e-06, "num_tokens": 124597.0, "completions/mean_length": 48.0, "completions/min_length": 40.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.0, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.8362926840782166, "rewards/meter/std": 0.15002521872520447, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9975625872612, "rewards/repeat_soft/std": 0.004202974960207939, "rewards/judge_quality/mean": 0.6150000095367432, "rewards/judge_quality/std": 0.25286927819252014, "rewards/total_composite/mean": 0.8105879426002502, "rewards/total_composite/std": 0.09124923497438431, "reward": 0.8105879426002502, "reward_std": 0.09124922752380371, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19722096621990204, "sampling/sampling_logp_difference/max": 4.355945110321045, "sampling/importance_sampling_ratio/min": 0.012830307707190514, "sampling/importance_sampling_ratio/mean": 1.0022335052490234, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8046837113797665, "clip_ratio/low_mean": 0.11510978080332279, "clip_ratio/low_min": 0.11510978080332279, "clip_ratio/high_mean": 0.06482380721718073, "clip_ratio/high_max": 0.06482380721718073, "clip_ratio/region_mean": 0.17993358802050352, "reward_total_mean": 0.8105879426002502, "reward_meter_mean": 0.8362926840782166, "reward_meter_std": 0.15002521872520447, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9975625872612, "reward_repeat_soft_std": 0.004202974960207939, "reward_judge_quality_mean": 0.6150000095367432, "reward_judge_quality_std": 0.25286927819252014, "reward_total_composite_mean": 0.8105879426002502, "reward_total_composite_std": 0.09124923497438431} {"timestamp_utc": "2026-04-12T22:23:09Z", "mode": "train", "global_step": 66, "epoch": 0.0066298342541436465, "loss": 0.7836, "grad_norm": 12.42125129699707, "learning_rate": 9.803030303030304e-06, "num_tokens": 126598.0, "completions/mean_length": 106.125, "completions/min_length": 35.0, "completions/max_length": 435.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.125, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 435.0, "rewards/meter/mean": 0.7745488882064819, "rewards/meter/std": 0.28584718704223633, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9980877637863159, "rewards/repeat_soft/std": 0.003999420441687107, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.21357084810733795, "rewards/total_composite/mean": 0.7292307615280151, "rewards/total_composite/std": 0.18889670073986053, "reward": 0.7292307615280151, "reward_std": 0.18889670073986053, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19706401228904724, "sampling/sampling_logp_difference/max": 1.9290742874145508, "sampling/importance_sampling_ratio/min": 0.14528262615203857, "sampling/importance_sampling_ratio/mean": 1.0292773246765137, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0778553932905197, "clip_ratio/low_mean": 0.0641922689974308, "clip_ratio/low_min": 0.0641922689974308, "clip_ratio/high_mean": 0.12423842307180166, "clip_ratio/high_max": 0.12423842307180166, "clip_ratio/region_mean": 0.18843069206923246, "reward_total_mean": 0.7292307615280151, "reward_meter_mean": 0.7745488882064819, "reward_meter_std": 0.28584718704223633, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9980877637863159, "reward_repeat_soft_std": 0.003999420441687107, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.21357084810733795, "reward_total_composite_mean": 0.7292307615280151, "reward_total_composite_std": 0.18889670073986053} {"timestamp_utc": "2026-04-12T22:23:17Z", "mode": "train", "global_step": 67, "epoch": 0.006730286288297338, "loss": -0.0138, "grad_norm": 12.676423072814941, "learning_rate": 9.800000000000001e-06, "num_tokens": 128362.0, "completions/mean_length": 61.5, "completions/min_length": 57.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.5, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9230592250823975, "rewards/meter/std": 0.17033740878105164, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.996062159538269, "rewards/repeat_soft/std": 0.007730784825980663, "rewards/judge_quality/mean": 0.5049999952316284, "rewards/judge_quality/std": 0.16801361739635468, "rewards/total_composite/mean": 0.8164828419685364, "rewards/total_composite/std": 0.10062293708324432, "reward": 0.8164828419685364, "reward_std": 0.10062292963266373, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16694454848766327, "sampling/sampling_logp_difference/max": 1.8855005502700806, "sampling/importance_sampling_ratio/min": 0.15175308287143707, "sampling/importance_sampling_ratio/mean": 1.0206559896469116, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.132050782442093, "clip_ratio/low_mean": 0.023706896230578423, "clip_ratio/low_min": 0.023706896230578423, "clip_ratio/high_mean": 0.15569907892495394, "clip_ratio/high_max": 0.15569907892495394, "clip_ratio/region_mean": 0.17940597515553236, "reward_total_mean": 0.8164828419685364, "reward_meter_mean": 0.9230592250823975, "reward_meter_std": 0.17033740878105164, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.996062159538269, "reward_repeat_soft_std": 0.007730784825980663, "reward_judge_quality_mean": 0.5049999952316284, "reward_judge_quality_std": 0.16801361739635468, "reward_total_composite_mean": 0.8164828419685364, "reward_total_composite_std": 0.10062293708324432} {"timestamp_utc": "2026-04-12T22:23:24Z", "mode": "train", "global_step": 68, "epoch": 0.00683073832245103, "loss": -0.0557, "grad_norm": 17.533666610717773, "learning_rate": 9.796969696969698e-06, "num_tokens": 129994.0, "completions/mean_length": 54.0, "completions/min_length": 37.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.29206711053848267, "rewards/meter/std": 0.2812570035457611, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9896067380905151, "rewards/repeat_soft/std": 0.01452327985316515, "rewards/judge_quality/mean": 0.4137499928474426, "rewards/judge_quality/std": 0.06781013309955597, "rewards/total_composite/mean": 0.4468875229358673, "rewards/total_composite/std": 0.21935208141803741, "reward": 0.4468875229358673, "reward_std": 0.21935206651687622, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24508477747440338, "sampling/sampling_logp_difference/max": 2.154849052429199, "sampling/importance_sampling_ratio/min": 0.11592069268226624, "sampling/importance_sampling_ratio/mean": 1.0226720571517944, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9609599560499191, "clip_ratio/low_mean": 0.08957079518586397, "clip_ratio/low_min": 0.08957079518586397, "clip_ratio/high_mean": 0.08130989596247673, "clip_ratio/high_max": 0.08130989596247673, "clip_ratio/region_mean": 0.1708806911483407, "reward_total_mean": 0.4468875229358673, "reward_meter_mean": 0.29206711053848267, "reward_meter_std": 0.2812570035457611, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9896067380905151, "reward_repeat_soft_std": 0.01452327985316515, "reward_judge_quality_mean": 0.4137499928474426, "reward_judge_quality_std": 0.06781013309955597, "reward_total_composite_mean": 0.4468875229358673, "reward_total_composite_std": 0.21935208141803741} {"timestamp_utc": "2026-04-12T22:23:36Z", "mode": "train", "global_step": 69, "epoch": 0.006931190356604721, "loss": 0.1552, "grad_norm": 18.79545783996582, "learning_rate": 9.793939393939394e-06, "num_tokens": 132069.0, "completions/mean_length": 72.375, "completions/min_length": 50.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.375, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.6283866167068481, "rewards/meter/std": 0.29875946044921875, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9875814914703369, "rewards/repeat_soft/std": 0.009496775455772877, "rewards/judge_quality/mean": 0.5199999809265137, "rewards/judge_quality/std": 0.19272483885288239, "rewards/total_composite/mean": 0.6875321865081787, "rewards/total_composite/std": 0.15890254080295563, "reward": 0.6875321865081787, "reward_std": 0.15890255570411682, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.26377734541893005, "sampling/sampling_logp_difference/max": 2.8466076850891113, "sampling/importance_sampling_ratio/min": 0.05804087966680527, "sampling/importance_sampling_ratio/mean": 1.0076905488967896, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4112783074378967, "clip_ratio/low_mean": 0.07913436740636826, "clip_ratio/low_min": 0.07913436740636826, "clip_ratio/high_mean": 0.1484571509063244, "clip_ratio/high_max": 0.1484571509063244, "clip_ratio/region_mean": 0.22759151831269264, "reward_total_mean": 0.6875321865081787, "reward_meter_mean": 0.6283866167068481, "reward_meter_std": 0.29875946044921875, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9875814914703369, "reward_repeat_soft_std": 0.009496775455772877, "reward_judge_quality_mean": 0.5199999809265137, "reward_judge_quality_std": 0.19272483885288239, "reward_total_composite_mean": 0.6875321865081787, "reward_total_composite_std": 0.15890254080295563} {"timestamp_utc": "2026-04-12T22:23:49Z", "mode": "train", "global_step": 70, "epoch": 0.007031642390758413, "loss": 0.0629, "grad_norm": 11.336478233337402, "learning_rate": 9.790909090909093e-06, "num_tokens": 134316.0, "completions/mean_length": 108.875, "completions/min_length": 50.0, "completions/max_length": 138.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.875, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 138.0, "rewards/meter/mean": 0.2835621237754822, "rewards/meter/std": 0.3008175492286682, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9945306777954102, "rewards/repeat_soft/std": 0.003017295151948929, "rewards/judge_quality/mean": 0.6362500190734863, "rewards/judge_quality/std": 0.24047201871871948, "rewards/total_composite/mean": 0.563243567943573, "rewards/total_composite/std": 0.17085498571395874, "reward": 0.563243567943573, "reward_std": 0.17085498571395874, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21989120543003082, "sampling/sampling_logp_difference/max": 1.4525012969970703, "sampling/importance_sampling_ratio/min": 0.23398429155349731, "sampling/importance_sampling_ratio/mean": 1.0469709634780884, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0914076268672943, "clip_ratio/low_mean": 0.18281028605997562, "clip_ratio/low_min": 0.18281028605997562, "clip_ratio/high_mean": 0.05735824815928936, "clip_ratio/high_max": 0.05735824815928936, "clip_ratio/region_mean": 0.24016853421926498, "reward_total_mean": 0.563243567943573, "reward_meter_mean": 0.2835621237754822, "reward_meter_std": 0.3008175492286682, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9945306777954102, "reward_repeat_soft_std": 0.003017295151948929, "reward_judge_quality_mean": 0.6362500190734863, "reward_judge_quality_std": 0.24047201871871948, "reward_total_composite_mean": 0.563243567943573, "reward_total_composite_std": 0.17085498571395874} {"timestamp_utc": "2026-04-12T22:23:58Z", "mode": "train", "global_step": 71, "epoch": 0.007132094424912104, "loss": 0.2256, "grad_norm": 17.786706924438477, "learning_rate": 9.787878787878788e-06, "num_tokens": 136031.0, "completions/mean_length": 42.375, "completions/min_length": 25.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.375, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.6308169364929199, "rewards/meter/std": 0.404389351606369, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9952167272567749, "rewards/repeat_soft/std": 0.004450792912393808, "rewards/judge_quality/mean": 0.7987500429153442, "rewards/judge_quality/std": 0.22465451061725616, "rewards/total_composite/mean": 0.7730143070220947, "rewards/total_composite/std": 0.22425441443920135, "reward": 0.7730143070220947, "reward_std": 0.22425441443920135, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15945900976657867, "sampling/sampling_logp_difference/max": 2.320220947265625, "sampling/importance_sampling_ratio/min": 0.09825187176465988, "sampling/importance_sampling_ratio/mean": 1.0230482816696167, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.073503103107214, "clip_ratio/low_mean": 0.07358044851571321, "clip_ratio/low_min": 0.07358044851571321, "clip_ratio/high_mean": 0.05085150431841612, "clip_ratio/high_max": 0.05085150431841612, "clip_ratio/region_mean": 0.12443195283412933, "reward_total_mean": 0.7730143070220947, "reward_meter_mean": 0.6308169364929199, "reward_meter_std": 0.404389351606369, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9952167272567749, "reward_repeat_soft_std": 0.004450792912393808, "reward_judge_quality_mean": 0.7987500429153442, "reward_judge_quality_std": 0.22465451061725616, "reward_total_composite_mean": 0.7730143070220947, "reward_total_composite_std": 0.22425441443920135} {"timestamp_utc": "2026-04-12T22:24:07Z", "mode": "train", "global_step": 72, "epoch": 0.007232546459065796, "loss": 0.0725, "grad_norm": 12.097882270812988, "learning_rate": 9.784848484848486e-06, "num_tokens": 138022.0, "completions/mean_length": 86.875, "completions/min_length": 77.0, "completions/max_length": 95.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.875, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.24359747767448425, "rewards/meter/std": 0.2615220844745636, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.993793249130249, "rewards/repeat_soft/std": 0.002918896032497287, "rewards/judge_quality/mean": 0.6112499833106995, "rewards/judge_quality/std": 0.15037456154823303, "rewards/total_composite/mean": 0.5423731803894043, "rewards/total_composite/std": 0.13441166281700134, "reward": 0.5423731803894043, "reward_std": 0.13441164791584015, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18156975507736206, "sampling/sampling_logp_difference/max": 3.1153430938720703, "sampling/importance_sampling_ratio/min": 0.04436328634619713, "sampling/importance_sampling_ratio/mean": 0.9998382329940796, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9134444296360016, "clip_ratio/low_mean": 0.1097878199070692, "clip_ratio/low_min": 0.1097878199070692, "clip_ratio/high_mean": 0.030120695009827614, "clip_ratio/high_max": 0.030120695009827614, "clip_ratio/region_mean": 0.13990851491689682, "reward_total_mean": 0.5423731803894043, "reward_meter_mean": 0.24359747767448425, "reward_meter_std": 0.2615220844745636, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.993793249130249, "reward_repeat_soft_std": 0.002918896032497287, "reward_judge_quality_mean": 0.6112499833106995, "reward_judge_quality_std": 0.15037456154823303, "reward_total_composite_mean": 0.5423731803894043, "reward_total_composite_std": 0.13441166281700134} {"timestamp_utc": "2026-04-12T22:24:19Z", "mode": "train", "global_step": 73, "epoch": 0.007332998493219488, "loss": 0.0614, "grad_norm": 18.151248931884766, "learning_rate": 9.781818181818183e-06, "num_tokens": 139596.0, "completions/mean_length": 37.75, "completions/min_length": 32.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.75, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.6459850072860718, "rewards/meter/std": 0.36253878474235535, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9962372779846191, "rewards/repeat_soft/std": 0.00649625901132822, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.6873170137405396, "rewards/total_composite/std": 0.13498029112815857, "reward": 0.6873170137405396, "reward_std": 0.13498029112815857, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20438161492347717, "sampling/sampling_logp_difference/max": 1.8445549011230469, "sampling/importance_sampling_ratio/min": 0.15809568762779236, "sampling/importance_sampling_ratio/mean": 1.0211987495422363, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.976985052227974, "clip_ratio/low_mean": 0.07190257450565696, "clip_ratio/low_min": 0.07190257450565696, "clip_ratio/high_mean": 0.08791267033666372, "clip_ratio/high_max": 0.08791267033666372, "clip_ratio/region_mean": 0.15981524484232068, "reward_total_mean": 0.6873170137405396, "reward_meter_mean": 0.6459850072860718, "reward_meter_std": 0.36253878474235535, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9962372779846191, "reward_repeat_soft_std": 0.00649625901132822, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.6873170137405396, "reward_total_composite_std": 0.13498029112815857} {"timestamp_utc": "2026-04-12T22:24:26Z", "mode": "train", "global_step": 74, "epoch": 0.007433450527373179, "loss": 0.0588, "grad_norm": 13.155890464782715, "learning_rate": 9.77878787878788e-06, "num_tokens": 141237.0, "completions/mean_length": 59.125, "completions/min_length": 51.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.125, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.6750097274780273, "rewards/meter/std": 0.36580297350883484, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9978744983673096, "rewards/repeat_soft/std": 0.0031460225582122803, "rewards/judge_quality/mean": 0.7437499761581421, "rewards/judge_quality/std": 0.33683136105537415, "rewards/total_composite/mean": 0.6564773321151733, "rewards/total_composite/std": 0.3408243656158447, "reward": 0.6564773321151733, "reward_std": 0.3408243656158447, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15842394530773163, "sampling/sampling_logp_difference/max": 1.5450174808502197, "sampling/importance_sampling_ratio/min": 0.2133081555366516, "sampling/importance_sampling_ratio/mean": 1.0020114183425903, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8815535008907318, "clip_ratio/low_mean": 0.04728598054498434, "clip_ratio/low_min": 0.04728598054498434, "clip_ratio/high_mean": 0.09232382103800774, "clip_ratio/high_max": 0.09232382103800774, "clip_ratio/region_mean": 0.13960980158299208, "reward_total_mean": 0.6564773321151733, "reward_meter_mean": 0.6750097274780273, "reward_meter_std": 0.36580297350883484, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9978744983673096, "reward_repeat_soft_std": 0.0031460225582122803, "reward_judge_quality_mean": 0.7437499761581421, "reward_judge_quality_std": 0.33683136105537415, "reward_total_composite_mean": 0.6564773321151733, "reward_total_composite_std": 0.3408243656158447} {"timestamp_utc": "2026-04-12T22:24:39Z", "mode": "train", "global_step": 75, "epoch": 0.007533902561526871, "loss": -0.0769, "grad_norm": 9.224102973937988, "learning_rate": 9.775757575757576e-06, "num_tokens": 143471.0, "completions/mean_length": 105.25, "completions/min_length": 74.0, "completions/max_length": 145.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 105.25, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.7617064714431763, "rewards/meter/std": 0.3298294246196747, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9947261214256287, "rewards/repeat_soft/std": 0.006881263107061386, "rewards/judge_quality/mean": 0.5449999570846558, "rewards/judge_quality/std": 0.1752549260854721, "rewards/total_composite/mean": 0.7557405233383179, "rewards/total_composite/std": 0.17117002606391907, "reward": 0.7557405233383179, "reward_std": 0.17117002606391907, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20651639997959137, "sampling/sampling_logp_difference/max": 1.6719465255737305, "sampling/importance_sampling_ratio/min": 0.1878809928894043, "sampling/importance_sampling_ratio/mean": 1.0183568000793457, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3894700407981873, "clip_ratio/low_mean": 0.07153031043708324, "clip_ratio/low_min": 0.07153031043708324, "clip_ratio/high_mean": 0.1301163136959076, "clip_ratio/high_max": 0.1301163136959076, "clip_ratio/region_mean": 0.20164662413299084, "reward_total_mean": 0.7557405233383179, "reward_meter_mean": 0.7617064714431763, "reward_meter_std": 0.3298294246196747, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9947261214256287, "reward_repeat_soft_std": 0.006881263107061386, "reward_judge_quality_mean": 0.5449999570846558, "reward_judge_quality_std": 0.1752549260854721, "reward_total_composite_mean": 0.7557405233383179, "reward_total_composite_std": 0.17117002606391907} {"timestamp_utc": "2026-04-12T22:24:49Z", "mode": "train", "global_step": 76, "epoch": 0.0076343545956805625, "loss": 0.0058, "grad_norm": 18.25312042236328, "learning_rate": 9.772727272727273e-06, "num_tokens": 144964.0, "completions/mean_length": 32.625, "completions/min_length": 25.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.625, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.708471417427063, "rewards/meter/std": 0.4398741126060486, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9486144185066223, "rewards/repeat_soft/std": 0.03010723739862442, "rewards/judge_quality/mean": 0.6150000095367432, "rewards/judge_quality/std": 0.25286927819252014, "rewards/total_composite/mean": 0.7481735944747925, "rewards/total_composite/std": 0.20565913617610931, "reward": 0.7481735944747925, "reward_std": 0.20565910637378693, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17192530632019043, "sampling/sampling_logp_difference/max": 0.9567670822143555, "sampling/importance_sampling_ratio/min": 0.3841327428817749, "sampling/importance_sampling_ratio/mean": 1.0192999839782715, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8063572198152542, "clip_ratio/low_mean": 0.06461021397262812, "clip_ratio/low_min": 0.06461021397262812, "clip_ratio/high_mean": 0.10132488049566746, "clip_ratio/high_max": 0.10132488049566746, "clip_ratio/region_mean": 0.16593509446829557, "reward_total_mean": 0.7481735944747925, "reward_meter_mean": 0.708471417427063, "reward_meter_std": 0.4398741126060486, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9486144185066223, "reward_repeat_soft_std": 0.03010723739862442, "reward_judge_quality_mean": 0.6150000095367432, "reward_judge_quality_std": 0.25286927819252014, "reward_total_composite_mean": 0.7481735944747925, "reward_total_composite_std": 0.20565913617610931} {"timestamp_utc": "2026-04-12T22:24:57Z", "mode": "train", "global_step": 77, "epoch": 0.0077348066298342545, "loss": -0.0728, "grad_norm": 17.3715763092041, "learning_rate": 9.76969696969697e-06, "num_tokens": 146633.0, "completions/mean_length": 52.625, "completions/min_length": 35.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.625, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.2402770221233368, "rewards/meter/std": 0.31088128685951233, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9913026094436646, "rewards/repeat_soft/std": 0.009482075460255146, "rewards/judge_quality/mean": 0.6862499713897705, "rewards/judge_quality/std": 0.28076112270355225, "rewards/total_composite/mean": 0.5631299018859863, "rewards/total_composite/std": 0.1694323569536209, "reward": 0.5631299018859863, "reward_std": 0.16943234205245972, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1446535289287567, "sampling/sampling_logp_difference/max": 1.915684700012207, "sampling/importance_sampling_ratio/min": 0.14724098145961761, "sampling/importance_sampling_ratio/mean": 1.030705451965332, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0392472222447395, "clip_ratio/low_mean": 0.11356967873871326, "clip_ratio/low_min": 0.11356967873871326, "clip_ratio/high_mean": 0.033929698169231415, "clip_ratio/high_max": 0.033929698169231415, "clip_ratio/region_mean": 0.14749937690794468, "reward_total_mean": 0.5631299018859863, "reward_meter_mean": 0.2402770221233368, "reward_meter_std": 0.31088128685951233, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9913026094436646, "reward_repeat_soft_std": 0.009482075460255146, "reward_judge_quality_mean": 0.6862499713897705, "reward_judge_quality_std": 0.28076112270355225, "reward_total_composite_mean": 0.5631299018859863, "reward_total_composite_std": 0.1694323569536209} {"timestamp_utc": "2026-04-12T22:25:05Z", "mode": "train", "global_step": 78, "epoch": 0.007835258663987946, "loss": 0.0969, "grad_norm": 9.234482765197754, "learning_rate": 9.766666666666667e-06, "num_tokens": 149203.0, "completions/mean_length": 126.25, "completions/min_length": 107.0, "completions/max_length": 153.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 126.25, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 153.0, "rewards/meter/mean": 0.9073135256767273, "rewards/meter/std": 0.11808735877275467, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9976153373718262, "rewards/repeat_soft/std": 0.0015691010048612952, "rewards/judge_quality/mean": 0.5487500429153442, "rewards/judge_quality/std": 0.23793382942676544, "rewards/total_composite/mean": 0.8226776123046875, "rewards/total_composite/std": 0.1000823825597763, "reward": 0.8226776123046875, "reward_std": 0.10008236765861511, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19733132421970367, "sampling/sampling_logp_difference/max": 1.7452077865600586, "sampling/importance_sampling_ratio/min": 0.17460869252681732, "sampling/importance_sampling_ratio/mean": 1.0099540948867798, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9953903406858444, "clip_ratio/low_mean": 0.11173281446099281, "clip_ratio/low_min": 0.11173281446099281, "clip_ratio/high_mean": 0.08277627266943455, "clip_ratio/high_max": 0.08277627266943455, "clip_ratio/region_mean": 0.19450908713042736, "reward_total_mean": 0.8226776123046875, "reward_meter_mean": 0.9073135256767273, "reward_meter_std": 0.11808735877275467, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9976153373718262, "reward_repeat_soft_std": 0.0015691010048612952, "reward_judge_quality_mean": 0.5487500429153442, "reward_judge_quality_std": 0.23793382942676544, "reward_total_composite_mean": 0.8226776123046875, "reward_total_composite_std": 0.1000823825597763} {"timestamp_utc": "2026-04-12T22:25:12Z", "mode": "train", "global_step": 79, "epoch": 0.007935710698141637, "loss": -0.0129, "grad_norm": 13.165565490722656, "learning_rate": 9.763636363636365e-06, "num_tokens": 151314.0, "completions/mean_length": 86.875, "completions/min_length": 71.0, "completions/max_length": 97.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.875, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.575776219367981, "rewards/meter/std": 0.40430617332458496, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9948314428329468, "rewards/repeat_soft/std": 0.004851632285863161, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.7033324241638184, "rewards/total_composite/std": 0.19417116045951843, "reward": 0.7033324241638184, "reward_std": 0.19417116045951843, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20565244555473328, "sampling/sampling_logp_difference/max": 1.9638829231262207, "sampling/importance_sampling_ratio/min": 0.1403125375509262, "sampling/importance_sampling_ratio/mean": 1.03525710105896, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8156342655420303, "clip_ratio/low_mean": 0.09965229965746403, "clip_ratio/low_min": 0.09965229965746403, "clip_ratio/high_mean": 0.10430697165429592, "clip_ratio/high_max": 0.10430697165429592, "clip_ratio/region_mean": 0.20395927131175995, "reward_total_mean": 0.7033324241638184, "reward_meter_mean": 0.575776219367981, "reward_meter_std": 0.40430617332458496, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9948314428329468, "reward_repeat_soft_std": 0.004851632285863161, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.7033324241638184, "reward_total_composite_std": 0.19417116045951843} {"timestamp_utc": "2026-04-12T22:25:18Z", "mode": "train", "global_step": 80, "epoch": 0.00803616273229533, "loss": -0.017, "grad_norm": 13.996071815490723, "learning_rate": 9.760606060606062e-06, "num_tokens": 153151.0, "completions/mean_length": 60.625, "completions/min_length": 48.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.625, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.786838948726654, "rewards/meter/std": 0.3647293746471405, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9986039400100708, "rewards/repeat_soft/std": 0.0022809384390711784, "rewards/judge_quality/mean": 0.4724999964237213, "rewards/judge_quality/std": 0.11310552060604095, "rewards/total_composite/mean": 0.7456879615783691, "rewards/total_composite/std": 0.13831521570682526, "reward": 0.7456879615783691, "reward_std": 0.13831521570682526, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21806959807872772, "sampling/sampling_logp_difference/max": 1.4953420162200928, "sampling/importance_sampling_ratio/min": 0.22417192161083221, "sampling/importance_sampling_ratio/mean": 1.045628547668457, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2772906571626663, "clip_ratio/low_mean": 0.05426747165620327, "clip_ratio/low_min": 0.05426747165620327, "clip_ratio/high_mean": 0.14767173863947392, "clip_ratio/high_max": 0.14767173863947392, "clip_ratio/region_mean": 0.20193921029567719, "reward_total_mean": 0.7456879615783691, "reward_meter_mean": 0.786838948726654, "reward_meter_std": 0.3647293746471405, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9986039400100708, "reward_repeat_soft_std": 0.0022809384390711784, "reward_judge_quality_mean": 0.4724999964237213, "reward_judge_quality_std": 0.11310552060604095, "reward_total_composite_mean": 0.7456879615783691, "reward_total_composite_std": 0.13831521570682526} {"timestamp_utc": "2026-04-12T22:25:28Z", "mode": "train", "global_step": 81, "epoch": 0.008136614766449021, "loss": 0.0024, "grad_norm": 10.145329475402832, "learning_rate": 9.757575757575758e-06, "num_tokens": 155821.0, "completions/mean_length": 123.75, "completions/min_length": 77.0, "completions/max_length": 148.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 123.75, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 148.0, "rewards/meter/mean": 0.8844622373580933, "rewards/meter/std": 0.30505692958831787, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9946202039718628, "rewards/repeat_soft/std": 0.009633982554078102, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.7800325155258179, "rewards/total_composite/std": 0.14473092555999756, "reward": 0.7800325155258179, "reward_std": 0.14473092555999756, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20969630777835846, "sampling/sampling_logp_difference/max": 2.1087703704833984, "sampling/importance_sampling_ratio/min": 0.12138713151216507, "sampling/importance_sampling_ratio/mean": 1.0255879163742065, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.578194409608841, "clip_ratio/low_mean": 0.024572649970650673, "clip_ratio/low_min": 0.024572649970650673, "clip_ratio/high_mean": 0.1784006580710411, "clip_ratio/high_max": 0.1784006580710411, "clip_ratio/region_mean": 0.20297330804169178, "reward_total_mean": 0.7800325155258179, "reward_meter_mean": 0.8844622373580933, "reward_meter_std": 0.30505692958831787, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9946202039718628, "reward_repeat_soft_std": 0.009633982554078102, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.7800325155258179, "reward_total_composite_std": 0.14473092555999756} {"timestamp_utc": "2026-04-12T22:25:36Z", "mode": "train", "global_step": 82, "epoch": 0.008237066800602712, "loss": 0.0932, "grad_norm": 15.969931602478027, "learning_rate": 9.754545454545455e-06, "num_tokens": 157487.0, "completions/mean_length": 59.25, "completions/min_length": 49.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.25, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.5042653679847717, "rewards/meter/std": 0.4921252727508545, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9970687627792358, "rewards/repeat_soft/std": 0.004391745664179325, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.6056262850761414, "rewards/total_composite/std": 0.26372030377388, "reward": 0.6056262850761414, "reward_std": 0.26372030377388, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21136270463466644, "sampling/sampling_logp_difference/max": 2.0872573852539062, "sampling/importance_sampling_ratio/min": 0.1240268275141716, "sampling/importance_sampling_ratio/mean": 1.0206705331802368, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.811131864786148, "clip_ratio/low_mean": 0.0915905861184001, "clip_ratio/low_min": 0.0915905861184001, "clip_ratio/high_mean": 0.09937021695077419, "clip_ratio/high_max": 0.09937021695077419, "clip_ratio/region_mean": 0.1909608030691743, "reward_total_mean": 0.6056262850761414, "reward_meter_mean": 0.5042653679847717, "reward_meter_std": 0.4921252727508545, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9970687627792358, "reward_repeat_soft_std": 0.004391745664179325, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.6056262850761414, "reward_total_composite_std": 0.26372030377388} {"timestamp_utc": "2026-04-12T22:25:43Z", "mode": "train", "global_step": 83, "epoch": 0.008337518834756403, "loss": 0.0093, "grad_norm": 13.831247329711914, "learning_rate": 9.751515151515152e-06, "num_tokens": 159104.0, "completions/mean_length": 57.125, "completions/min_length": 40.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.125, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.7920023202896118, "rewards/meter/std": 0.3581661581993103, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9975086450576782, "rewards/repeat_soft/std": 0.003981848247349262, "rewards/judge_quality/mean": 0.5862500071525574, "rewards/judge_quality/std": 0.2822834253311157, "rewards/total_composite/mean": 0.7820268869400024, "rewards/total_composite/std": 0.22069051861763, "reward": 0.7820268869400024, "reward_std": 0.22069051861763, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18509216606616974, "sampling/sampling_logp_difference/max": 1.5339620113372803, "sampling/importance_sampling_ratio/min": 0.21567945182323456, "sampling/importance_sampling_ratio/mean": 1.0375295877456665, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7064232379198074, "clip_ratio/low_mean": 0.044791667722165585, "clip_ratio/low_min": 0.044791667722165585, "clip_ratio/high_mean": 0.13707098085433245, "clip_ratio/high_max": 0.13707098085433245, "clip_ratio/region_mean": 0.18186264857649803, "reward_total_mean": 0.7820268869400024, "reward_meter_mean": 0.7920023202896118, "reward_meter_std": 0.3581661581993103, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9975086450576782, "reward_repeat_soft_std": 0.003981848247349262, "reward_judge_quality_mean": 0.5862500071525574, "reward_judge_quality_std": 0.2822834253311157, "reward_total_composite_mean": 0.7820268869400024, "reward_total_composite_std": 0.22069051861763} {"timestamp_utc": "2026-04-12T22:25:51Z", "mode": "train", "global_step": 84, "epoch": 0.008437970868910096, "loss": -0.0464, "grad_norm": 18.266530990600586, "learning_rate": 9.74848484848485e-06, "num_tokens": 160981.0, "completions/mean_length": 53.625, "completions/min_length": 38.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.625, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.674973726272583, "rewards/meter/std": 0.4194721281528473, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9981741905212402, "rewards/repeat_soft/std": 0.0017913100309669971, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.682930588722229, "rewards/total_composite/std": 0.18727268278598785, "reward": 0.682930588722229, "reward_std": 0.18727268278598785, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2229832410812378, "sampling/sampling_logp_difference/max": 1.926133155822754, "sampling/importance_sampling_ratio/min": 0.1657637655735016, "sampling/importance_sampling_ratio/mean": 1.0519226789474487, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2065766006708145, "clip_ratio/low_mean": 0.07867324538528919, "clip_ratio/low_min": 0.07867324538528919, "clip_ratio/high_mean": 0.11280264891684055, "clip_ratio/high_max": 0.11280264891684055, "clip_ratio/region_mean": 0.19147589430212975, "reward_total_mean": 0.682930588722229, "reward_meter_mean": 0.674973726272583, "reward_meter_std": 0.4194721281528473, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9981741905212402, "reward_repeat_soft_std": 0.0017913100309669971, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.682930588722229, "reward_total_composite_std": 0.18727268278598785} {"timestamp_utc": "2026-04-12T22:25:57Z", "mode": "train", "global_step": 85, "epoch": 0.008538422903063787, "loss": 0.1016, "grad_norm": 26.06890106201172, "learning_rate": 9.745454545454547e-06, "num_tokens": 162397.0, "completions/mean_length": 32.0, "completions/min_length": 15.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.5776013731956482, "rewards/meter/std": 0.4645298421382904, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9671874642372131, "rewards/repeat_soft/std": 0.013258260674774647, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.5890144109725952, "rewards/total_composite/std": 0.3000996708869934, "reward": 0.5890144109725952, "reward_std": 0.3000996708869934, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2313683032989502, "sampling/sampling_logp_difference/max": 1.7472820281982422, "sampling/importance_sampling_ratio/min": 0.1742469072341919, "sampling/importance_sampling_ratio/mean": 1.0024120807647705, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7302970439195633, "clip_ratio/low_mean": 0.0708547318354249, "clip_ratio/low_min": 0.0708547318354249, "clip_ratio/high_mean": 0.1231324952095747, "clip_ratio/high_max": 0.1231324952095747, "clip_ratio/region_mean": 0.1939872270449996, "reward_total_mean": 0.5890144109725952, "reward_meter_mean": 0.5776013731956482, "reward_meter_std": 0.4645298421382904, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9671874642372131, "reward_repeat_soft_std": 0.013258260674774647, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.5890144109725952, "reward_total_composite_std": 0.3000996708869934} {"timestamp_utc": "2026-04-12T22:26:04Z", "mode": "train", "global_step": 86, "epoch": 0.008638874937217478, "loss": 0.0885, "grad_norm": 19.318851470947266, "learning_rate": 9.742424242424244e-06, "num_tokens": 164496.0, "completions/mean_length": 83.375, "completions/min_length": 70.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 83.375, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.5039986968040466, "rewards/meter/std": 0.3009590804576874, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9974728226661682, "rewards/repeat_soft/std": 0.0023224493488669395, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.5930466651916504, "rewards/total_composite/std": 0.14554795622825623, "reward": 0.5930466651916504, "reward_std": 0.14554795622825623, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.234708771109581, "sampling/sampling_logp_difference/max": 2.4151086807250977, "sampling/importance_sampling_ratio/min": 0.08935762941837311, "sampling/importance_sampling_ratio/mean": 0.986754834651947, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1506693363189697, "clip_ratio/low_mean": 0.07777538895606995, "clip_ratio/low_min": 0.07777538895606995, "clip_ratio/high_mean": 0.1278723981231451, "clip_ratio/high_max": 0.1278723981231451, "clip_ratio/region_mean": 0.20564778707921505, "reward_total_mean": 0.5930466651916504, "reward_meter_mean": 0.5039986968040466, "reward_meter_std": 0.3009590804576874, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9974728226661682, "reward_repeat_soft_std": 0.0023224493488669395, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.5930466651916504, "reward_total_composite_std": 0.14554795622825623} {"timestamp_utc": "2026-04-12T22:26:11Z", "mode": "train", "global_step": 87, "epoch": 0.00873932697137117, "loss": 0.047, "grad_norm": 32.492706298828125, "learning_rate": 9.739393939393941e-06, "num_tokens": 165854.0, "completions/mean_length": 28.75, "completions/min_length": 25.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.75, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.8606054782867432, "rewards/meter/std": 0.29568642377853394, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9594065546989441, "rewards/repeat_soft/std": 0.008749538101255894, "rewards/judge_quality/mean": 0.6150000095367432, "rewards/judge_quality/std": 0.25286927819252014, "rewards/total_composite/mean": 0.8177131414413452, "rewards/total_composite/std": 0.17395693063735962, "reward": 0.8177131414413452, "reward_std": 0.17395691573619843, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1784212738275528, "sampling/sampling_logp_difference/max": 1.7252826690673828, "sampling/importance_sampling_ratio/min": 0.17812269926071167, "sampling/importance_sampling_ratio/mean": 1.010380744934082, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4293856024742126, "clip_ratio/low_mean": 0.07695782743394375, "clip_ratio/low_min": 0.07695782743394375, "clip_ratio/high_mean": 0.11240384820848703, "clip_ratio/high_max": 0.11240384820848703, "clip_ratio/region_mean": 0.18936167564243078, "reward_total_mean": 0.8177131414413452, "reward_meter_mean": 0.8606054782867432, "reward_meter_std": 0.29568642377853394, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9594065546989441, "reward_repeat_soft_std": 0.008749538101255894, "reward_judge_quality_mean": 0.6150000095367432, "reward_judge_quality_std": 0.25286927819252014, "reward_total_composite_mean": 0.8177131414413452, "reward_total_composite_std": 0.17395693063735962} {"timestamp_utc": "2026-04-12T22:26:17Z", "mode": "train", "global_step": 88, "epoch": 0.008839779005524863, "loss": 0.0314, "grad_norm": 15.27247142791748, "learning_rate": 9.736363636363637e-06, "num_tokens": 167652.0, "completions/mean_length": 63.75, "completions/min_length": 62.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.75, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.6508787870407104, "rewards/meter/std": 0.44921696186065674, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9988118410110474, "rewards/repeat_soft/std": 0.0022364838514477015, "rewards/judge_quality/mean": 0.5387500524520874, "rewards/judge_quality/std": 0.24485784769058228, "rewards/total_composite/mean": 0.7044016122817993, "rewards/total_composite/std": 0.15757334232330322, "reward": 0.7044016122817993, "reward_std": 0.15757334232330322, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20896635949611664, "sampling/sampling_logp_difference/max": 1.6069178581237793, "sampling/importance_sampling_ratio/min": 0.20050464570522308, "sampling/importance_sampling_ratio/mean": 1.016836166381836, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5759567469358444, "clip_ratio/low_mean": 0.0732837338000536, "clip_ratio/low_min": 0.0732837338000536, "clip_ratio/high_mean": 0.13178288005292416, "clip_ratio/high_max": 0.13178288005292416, "clip_ratio/region_mean": 0.20506661385297775, "reward_total_mean": 0.7044016122817993, "reward_meter_mean": 0.6508787870407104, "reward_meter_std": 0.44921696186065674, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9988118410110474, "reward_repeat_soft_std": 0.0022364838514477015, "reward_judge_quality_mean": 0.5387500524520874, "reward_judge_quality_std": 0.24485784769058228, "reward_total_composite_mean": 0.7044016122817993, "reward_total_composite_std": 0.15757334232330322} {"timestamp_utc": "2026-04-12T22:26:24Z", "mode": "train", "global_step": 89, "epoch": 0.008940231039678554, "loss": 0.0346, "grad_norm": 14.53501033782959, "learning_rate": 9.733333333333334e-06, "num_tokens": 169728.0, "completions/mean_length": 94.5, "completions/min_length": 91.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.5, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.7820674180984497, "rewards/meter/std": 0.24537409842014313, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9976636171340942, "rewards/repeat_soft/std": 0.0012526443460956216, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.765196681022644, "rewards/total_composite/std": 0.10896959155797958, "reward": 0.765196681022644, "reward_std": 0.10896960645914078, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14851930737495422, "sampling/sampling_logp_difference/max": 1.868389368057251, "sampling/importance_sampling_ratio/min": 0.15437209606170654, "sampling/importance_sampling_ratio/mean": 1.010309100151062, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7215096801519394, "clip_ratio/low_mean": 0.03587029594928026, "clip_ratio/low_min": 0.03587029594928026, "clip_ratio/high_mean": 0.09394399169832468, "clip_ratio/high_max": 0.09394399169832468, "clip_ratio/region_mean": 0.12981428764760494, "reward_total_mean": 0.765196681022644, "reward_meter_mean": 0.7820674180984497, "reward_meter_std": 0.24537409842014313, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9976636171340942, "reward_repeat_soft_std": 0.0012526443460956216, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.765196681022644, "reward_total_composite_std": 0.10896959155797958} {"timestamp_utc": "2026-04-12T22:26:31Z", "mode": "train", "global_step": 90, "epoch": 0.009040683073832245, "loss": 0.0817, "grad_norm": 29.59337043762207, "learning_rate": 9.730303030303031e-06, "num_tokens": 171248.0, "completions/mean_length": 33.0, "completions/min_length": 21.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.0, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.418997198343277, "rewards/meter/std": 0.3585245907306671, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9966121912002563, "rewards/repeat_soft/std": 0.00797121599316597, "rewards/judge_quality/mean": 0.53125, "rewards/judge_quality/std": 0.1865811049938202, "rewards/total_composite/mean": 0.5975849628448486, "rewards/total_composite/std": 0.19449637830257416, "reward": 0.5975849628448486, "reward_std": 0.19449637830257416, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.25027692317962646, "sampling/sampling_logp_difference/max": 2.5391225814819336, "sampling/importance_sampling_ratio/min": 0.07893563061952591, "sampling/importance_sampling_ratio/mean": 0.9951542019844055, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5199098736047745, "clip_ratio/low_mean": 0.09931824635714293, "clip_ratio/low_min": 0.09931824635714293, "clip_ratio/high_mean": 0.11300505325198174, "clip_ratio/high_max": 0.11300505325198174, "clip_ratio/region_mean": 0.21232329960912466, "reward_total_mean": 0.5975849628448486, "reward_meter_mean": 0.418997198343277, "reward_meter_std": 0.3585245907306671, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9966121912002563, "reward_repeat_soft_std": 0.00797121599316597, "reward_judge_quality_mean": 0.53125, "reward_judge_quality_std": 0.1865811049938202, "reward_total_composite_mean": 0.5975849628448486, "reward_total_composite_std": 0.19449637830257416} {"timestamp_utc": "2026-04-12T22:26:39Z", "mode": "train", "global_step": 91, "epoch": 0.009141135107985936, "loss": 0.1256, "grad_norm": 10.04101848602295, "learning_rate": 9.727272727272728e-06, "num_tokens": 173561.0, "completions/mean_length": 116.125, "completions/min_length": 95.0, "completions/max_length": 150.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 116.125, "completions/min_terminated_length": 95.0, "completions/max_terminated_length": 150.0, "rewards/meter/mean": 0.693138837814331, "rewards/meter/std": 0.36694416403770447, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9948671460151672, "rewards/repeat_soft/std": 0.005335790105164051, "rewards/judge_quality/mean": 0.4737499952316284, "rewards/judge_quality/std": 0.16291432082653046, "rewards/total_composite/mean": 0.6941492557525635, "rewards/total_composite/std": 0.18267148733139038, "reward": 0.6941492557525635, "reward_std": 0.18267148733139038, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21353013813495636, "sampling/sampling_logp_difference/max": 2.810418128967285, "sampling/importance_sampling_ratio/min": 0.060179825872182846, "sampling/importance_sampling_ratio/mean": 1.0399506092071533, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0453978031873703, "clip_ratio/low_mean": 0.05563131347298622, "clip_ratio/low_min": 0.05563131347298622, "clip_ratio/high_mean": 0.15723679400980473, "clip_ratio/high_max": 0.15723679400980473, "clip_ratio/region_mean": 0.21286810748279095, "reward_total_mean": 0.6941492557525635, "reward_meter_mean": 0.693138837814331, "reward_meter_std": 0.36694416403770447, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9948671460151672, "reward_repeat_soft_std": 0.005335790105164051, "reward_judge_quality_mean": 0.4737499952316284, "reward_judge_quality_std": 0.16291432082653046, "reward_total_composite_mean": 0.6941492557525635, "reward_total_composite_std": 0.18267148733139038} {"timestamp_utc": "2026-04-12T22:26:47Z", "mode": "train", "global_step": 92, "epoch": 0.009241587142139629, "loss": 0.0532, "grad_norm": 14.25716781616211, "learning_rate": 9.724242424242426e-06, "num_tokens": 175372.0, "completions/mean_length": 58.375, "completions/min_length": 41.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.375, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.26952993869781494, "rewards/meter/std": 0.3523905575275421, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9935294389724731, "rewards/repeat_soft/std": 0.008154787123203278, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.500016450881958, "rewards/total_composite/std": 0.15747496485710144, "reward": 0.500016450881958, "reward_std": 0.15747497975826263, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.218789741396904, "sampling/sampling_logp_difference/max": 1.3333110809326172, "sampling/importance_sampling_ratio/min": 0.2674177289009094, "sampling/importance_sampling_ratio/mean": 1.0159591436386108, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6216311752796173, "clip_ratio/low_mean": 0.1285328259691596, "clip_ratio/low_min": 0.1285328259691596, "clip_ratio/high_mean": 0.06342516466975212, "clip_ratio/high_max": 0.06342516466975212, "clip_ratio/region_mean": 0.19195799063891172, "reward_total_mean": 0.500016450881958, "reward_meter_mean": 0.26952993869781494, "reward_meter_std": 0.3523905575275421, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9935294389724731, "reward_repeat_soft_std": 0.008154787123203278, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.500016450881958, "reward_total_composite_std": 0.15747496485710144} {"timestamp_utc": "2026-04-12T22:26:54Z", "mode": "train", "global_step": 93, "epoch": 0.00934203917629332, "loss": 0.0745, "grad_norm": 10.899441719055176, "learning_rate": 9.721212121212123e-06, "num_tokens": 177106.0, "completions/mean_length": 69.75, "completions/min_length": 53.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.75, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9947660565376282, "rewards/meter/std": 0.0041707707569003105, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9986412525177002, "rewards/repeat_soft/std": 0.002481638453900814, "rewards/judge_quality/mean": 0.3725000023841858, "rewards/judge_quality/std": 0.11055056750774384, "rewards/total_composite/mean": 0.8092588186264038, "rewards/total_composite/std": 0.033664677292108536, "reward": 0.8092588186264038, "reward_std": 0.033664681017398834, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17616569995880127, "sampling/sampling_logp_difference/max": 1.2230968475341797, "sampling/importance_sampling_ratio/min": 0.2943173050880432, "sampling/importance_sampling_ratio/mean": 1.0400642156600952, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3353325724601746, "clip_ratio/low_mean": 0.038446055725216866, "clip_ratio/low_min": 0.038446055725216866, "clip_ratio/high_mean": 0.14749629236757755, "clip_ratio/high_max": 0.14749629236757755, "clip_ratio/region_mean": 0.18594234809279442, "reward_total_mean": 0.8092588186264038, "reward_meter_mean": 0.9947660565376282, "reward_meter_std": 0.0041707707569003105, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9986412525177002, "reward_repeat_soft_std": 0.002481638453900814, "reward_judge_quality_mean": 0.3725000023841858, "reward_judge_quality_std": 0.11055056750774384, "reward_total_composite_mean": 0.8092588186264038, "reward_total_composite_std": 0.033664677292108536} {"timestamp_utc": "2026-04-12T22:27:00Z", "mode": "train", "global_step": 94, "epoch": 0.009442491210447011, "loss": 0.0462, "grad_norm": 17.971059799194336, "learning_rate": 9.718181818181818e-06, "num_tokens": 178827.0, "completions/mean_length": 37.125, "completions/min_length": 32.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.125, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.41076311469078064, "rewards/meter/std": 0.3964845538139343, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9856504797935486, "rewards/repeat_soft/std": 0.010154510848224163, "rewards/judge_quality/mean": 0.6775000095367432, "rewards/judge_quality/std": 0.25949129462242126, "rewards/total_composite/mean": 0.6366584300994873, "rewards/total_composite/std": 0.15718963742256165, "reward": 0.6366584300994873, "reward_std": 0.15718962252140045, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19394990801811218, "sampling/sampling_logp_difference/max": 1.2942476272583008, "sampling/importance_sampling_ratio/min": 0.2741039991378784, "sampling/importance_sampling_ratio/mean": 1.0088485479354858, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3612593337893486, "clip_ratio/low_mean": 0.08777131512761116, "clip_ratio/low_min": 0.08777131512761116, "clip_ratio/high_mean": 0.07181490398943424, "clip_ratio/high_max": 0.07181490398943424, "clip_ratio/region_mean": 0.1595862191170454, "reward_total_mean": 0.6366584300994873, "reward_meter_mean": 0.41076311469078064, "reward_meter_std": 0.3964845538139343, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9856504797935486, "reward_repeat_soft_std": 0.010154510848224163, "reward_judge_quality_mean": 0.6775000095367432, "reward_judge_quality_std": 0.25949129462242126, "reward_total_composite_mean": 0.6366584300994873, "reward_total_composite_std": 0.15718963742256165} {"timestamp_utc": "2026-04-12T22:27:08Z", "mode": "train", "global_step": 95, "epoch": 0.009542943244600702, "loss": 0.0122, "grad_norm": 10.587267875671387, "learning_rate": 9.715151515151516e-06, "num_tokens": 181305.0, "completions/mean_length": 126.75, "completions/min_length": 107.0, "completions/max_length": 151.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 126.75, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 151.0, "rewards/meter/mean": 0.9200213551521301, "rewards/meter/std": 0.1255701780319214, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9956197738647461, "rewards/repeat_soft/std": 0.003646343480795622, "rewards/judge_quality/mean": 0.6150000095367432, "rewards/judge_quality/std": 0.21293865144252777, "rewards/total_composite/mean": 0.8480715751647949, "rewards/total_composite/std": 0.08985444903373718, "reward": 0.8480715751647949, "reward_std": 0.08985444158315659, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18390880525112152, "sampling/sampling_logp_difference/max": 1.2477107048034668, "sampling/importance_sampling_ratio/min": 0.2871614396572113, "sampling/importance_sampling_ratio/mean": 1.0285959243774414, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9829654097557068, "clip_ratio/low_mean": 0.08566484786570072, "clip_ratio/low_min": 0.08566484786570072, "clip_ratio/high_mean": 0.09545699134469032, "clip_ratio/high_max": 0.09545699134469032, "clip_ratio/region_mean": 0.18112183921039104, "reward_total_mean": 0.8480715751647949, "reward_meter_mean": 0.9200213551521301, "reward_meter_std": 0.1255701780319214, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9956197738647461, "reward_repeat_soft_std": 0.003646343480795622, "reward_judge_quality_mean": 0.6150000095367432, "reward_judge_quality_std": 0.21293865144252777, "reward_total_composite_mean": 0.8480715751647949, "reward_total_composite_std": 0.08985444903373718} {"timestamp_utc": "2026-04-12T22:27:15Z", "mode": "train", "global_step": 96, "epoch": 0.009643395278754395, "loss": 0.1148, "grad_norm": 14.349672317504883, "learning_rate": 9.712121212121213e-06, "num_tokens": 183026.0, "completions/mean_length": 60.125, "completions/min_length": 41.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.125, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.37904417514801025, "rewards/meter/std": 0.36078208684921265, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9731553792953491, "rewards/repeat_soft/std": 0.04177224636077881, "rewards/judge_quality/mean": 0.5487500429153442, "rewards/judge_quality/std": 0.22937415540218353, "rewards/total_composite/mean": 0.5731354355812073, "rewards/total_composite/std": 0.18765978515148163, "reward": 0.5731354355812073, "reward_std": 0.18765978515148163, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20568451285362244, "sampling/sampling_logp_difference/max": 1.8774585723876953, "sampling/importance_sampling_ratio/min": 0.15297840535640717, "sampling/importance_sampling_ratio/mean": 1.0121736526489258, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8408035784959793, "clip_ratio/low_mean": 0.09115871600806713, "clip_ratio/low_min": 0.09115871600806713, "clip_ratio/high_mean": 0.10624979063868523, "clip_ratio/high_max": 0.10624979063868523, "clip_ratio/region_mean": 0.19740850664675236, "reward_total_mean": 0.5731354355812073, "reward_meter_mean": 0.37904417514801025, "reward_meter_std": 0.36078208684921265, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9731553792953491, "reward_repeat_soft_std": 0.04177224636077881, "reward_judge_quality_mean": 0.5487500429153442, "reward_judge_quality_std": 0.22937415540218353, "reward_total_composite_mean": 0.5731354355812073, "reward_total_composite_std": 0.18765978515148163} {"timestamp_utc": "2026-04-12T22:27:21Z", "mode": "train", "global_step": 97, "epoch": 0.009743847312908087, "loss": -0.0658, "grad_norm": 13.329829216003418, "learning_rate": 9.70909090909091e-06, "num_tokens": 184785.0, "completions/mean_length": 54.875, "completions/min_length": 35.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.875, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.44293856620788574, "rewards/meter/std": 0.3831910789012909, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9962660074234009, "rewards/repeat_soft/std": 0.004540997091680765, "rewards/judge_quality/mean": 0.38874998688697815, "rewards/judge_quality/std": 0.08675704896450043, "rewards/total_composite/mean": 0.4797763526439667, "rewards/total_composite/std": 0.24937774240970612, "reward": 0.4797763526439667, "reward_std": 0.24937772750854492, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18477389216423035, "sampling/sampling_logp_difference/max": 1.34785795211792, "sampling/importance_sampling_ratio/min": 0.2597961723804474, "sampling/importance_sampling_ratio/mean": 1.0322825908660889, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9412241727113724, "clip_ratio/low_mean": 0.07615137845277786, "clip_ratio/low_min": 0.07615137845277786, "clip_ratio/high_mean": 0.08887524623423815, "clip_ratio/high_max": 0.08887524623423815, "clip_ratio/region_mean": 0.165026624687016, "reward_total_mean": 0.4797763526439667, "reward_meter_mean": 0.44293856620788574, "reward_meter_std": 0.3831910789012909, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9962660074234009, "reward_repeat_soft_std": 0.004540997091680765, "reward_judge_quality_mean": 0.38874998688697815, "reward_judge_quality_std": 0.08675704896450043, "reward_total_composite_mean": 0.4797763526439667, "reward_total_composite_std": 0.24937774240970612} {"timestamp_utc": "2026-04-12T22:27:28Z", "mode": "train", "global_step": 98, "epoch": 0.009844299347061778, "loss": 0.0294, "grad_norm": 10.771571159362793, "learning_rate": 9.706060606060606e-06, "num_tokens": 186626.0, "completions/mean_length": 66.125, "completions/min_length": 57.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.125, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.6990766525268555, "rewards/meter/std": 0.38784509897232056, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9952908754348755, "rewards/repeat_soft/std": 0.007525984197854996, "rewards/judge_quality/mean": 0.4762499928474426, "rewards/judge_quality/std": 0.19167961180210114, "rewards/total_composite/mean": 0.7069886326789856, "rewards/total_composite/std": 0.19634383916854858, "reward": 0.7069886326789856, "reward_std": 0.19634383916854858, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19846424460411072, "sampling/sampling_logp_difference/max": 1.7335278987884521, "sampling/importance_sampling_ratio/min": 0.17666007578372955, "sampling/importance_sampling_ratio/mean": 1.0128724575042725, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7142320424318314, "clip_ratio/low_mean": 0.04686454962939024, "clip_ratio/low_min": 0.04686454962939024, "clip_ratio/high_mean": 0.12352950684726238, "clip_ratio/high_max": 0.12352950684726238, "clip_ratio/region_mean": 0.17039405647665262, "reward_total_mean": 0.7069886326789856, "reward_meter_mean": 0.6990766525268555, "reward_meter_std": 0.38784509897232056, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9952908754348755, "reward_repeat_soft_std": 0.007525984197854996, "reward_judge_quality_mean": 0.4762499928474426, "reward_judge_quality_std": 0.19167961180210114, "reward_total_composite_mean": 0.7069886326789856, "reward_total_composite_std": 0.19634383916854858} {"timestamp_utc": "2026-04-12T22:27:35Z", "mode": "train", "global_step": 99, "epoch": 0.009944751381215469, "loss": 0.0833, "grad_norm": 17.894399642944336, "learning_rate": 9.703030303030305e-06, "num_tokens": 188002.0, "completions/mean_length": 28.0, "completions/min_length": 21.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.0, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.49489158391952515, "rewards/meter/std": 0.4102099537849426, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9463370442390442, "rewards/repeat_soft/std": 0.02908223308622837, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.5978349447250366, "rewards/total_composite/std": 0.18156224489212036, "reward": 0.5978349447250366, "reward_std": 0.18156225979328156, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.172475203871727, "sampling/sampling_logp_difference/max": 1.023299217224121, "sampling/importance_sampling_ratio/min": 0.3594072461128235, "sampling/importance_sampling_ratio/mean": 1.049785852432251, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2047024965286255, "clip_ratio/low_mean": 0.08318580687046051, "clip_ratio/low_min": 0.08318580687046051, "clip_ratio/high_mean": 0.07235023286193609, "clip_ratio/high_max": 0.07235023286193609, "clip_ratio/region_mean": 0.1555360397323966, "reward_total_mean": 0.5978349447250366, "reward_meter_mean": 0.49489158391952515, "reward_meter_std": 0.4102099537849426, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9463370442390442, "reward_repeat_soft_std": 0.02908223308622837, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.5978349447250366, "reward_total_composite_std": 0.18156224489212036} {"timestamp_utc": "2026-04-12T22:27:42Z", "mode": "train", "global_step": 100, "epoch": 0.010045203415369162, "loss": 0.0883, "grad_norm": 10.852298736572266, "learning_rate": 9.7e-06, "num_tokens": 190092.0, "completions/mean_length": 87.25, "completions/min_length": 70.0, "completions/max_length": 100.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.25, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 100.0, "rewards/meter/mean": 0.25189706683158875, "rewards/meter/std": 0.2712648808956146, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9939076900482178, "rewards/repeat_soft/std": 0.004518185276538134, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.5112444162368774, "rewards/total_composite/std": 0.12693306803703308, "reward": 0.5112444162368774, "reward_std": 0.1269330531358719, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21233618259429932, "sampling/sampling_logp_difference/max": 1.5831584930419922, "sampling/importance_sampling_ratio/min": 0.20532555878162384, "sampling/importance_sampling_ratio/mean": 1.0322673320770264, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8813354671001434, "clip_ratio/low_mean": 0.14453337900340557, "clip_ratio/low_min": 0.14453337900340557, "clip_ratio/high_mean": 0.053007518872618675, "clip_ratio/high_max": 0.053007518872618675, "clip_ratio/region_mean": 0.19754089787602425, "reward_total_mean": 0.5112444162368774, "reward_meter_mean": 0.25189706683158875, "reward_meter_std": 0.2712648808956146, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9939076900482178, "reward_repeat_soft_std": 0.004518185276538134, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.5112444162368774, "reward_total_composite_std": 0.12693306803703308} {"timestamp_utc": "2026-04-12T22:28:35Z", "mode": "eval", "global_step": 100, "epoch": 0.010045203415369162, "eval_loss": NaN, "eval_runtime": 53.4733, "eval_samples_per_second": 1.496, "eval_steps_per_second": 0.187, "eval_num_tokens": 190092.0, "eval_completions/mean_length": 89.825, "eval_completions/min_length": 35.7, "eval_completions/max_length": 160.4, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 89.825, "eval_completions/min_terminated_length": 35.7, "eval_completions/max_terminated_length": 160.4, "eval_rewards/meter/mean": 0.5703477174043655, "eval_rewards/meter/std": 0.38051448166370394, "eval_rewards/count_adherence/mean": 0.9891666650772095, "eval_rewards/count_adherence/std": 0.025364020839333534, "eval_rewards/hard_gate/mean": 0.9125, "eval_rewards/hard_gate/std": 0.2230676978826523, "eval_rewards/repeat_soft/mean": 0.9923222184181213, "eval_rewards/repeat_soft/std": 0.011972753837471827, "eval_rewards/judge_quality/mean": 0.48449999690055845, "eval_rewards/judge_quality/std": 0.16168474704027175, "eval_rewards/total_composite/mean": 0.595814323425293, "eval_rewards/total_composite/std": 0.2441583454608917, "eval_reward": 0.595814323425293, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.13042871057987213, "eval_sampling/sampling_logp_difference/max": 1.1281907081604003, "eval_sampling/importance_sampling_ratio/min": 0.3298598065972328, "eval_sampling/importance_sampling_ratio/mean": 1.0367148160934447, "eval_sampling/importance_sampling_ratio/max": 1.5091773986816406, "eval_entropy": 1.9399710893630981, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.595814323425293, "eval_reward_meter_mean": 0.5703477174043655, "eval_reward_meter_std": 0.38051448166370394, "eval_reward_count_adherence_mean": 0.9891666650772095, "eval_reward_count_adherence_std": 0.025364020839333534, "eval_reward_hard_gate_mean": 0.9125, "eval_reward_hard_gate_std": 0.2230676978826523, "eval_reward_repeat_soft_mean": 0.9923222184181213, "eval_reward_repeat_soft_std": 0.011972753837471827, "eval_reward_judge_quality_mean": 0.48449999690055845, "eval_reward_judge_quality_std": 0.16168474704027175, "eval_reward_total_composite_mean": 0.595814323425293, "eval_reward_total_composite_std": 0.2441583454608917} {"timestamp_utc": "2026-04-12T22:28:46Z", "mode": "train", "global_step": 101, "epoch": 0.010145655449522853, "loss": 0.0307, "grad_norm": 8.720218658447266, "learning_rate": 9.696969696969698e-06, "num_tokens": 192271.0, "completions/mean_length": 96.375, "completions/min_length": 83.0, "completions/max_length": 104.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.375, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.9825379252433777, "rewards/meter/std": 0.03238074481487274, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9823436737060547, "rewards/repeat_soft/std": 0.021364767104387283, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.8100014328956604, "rewards/total_composite/std": 0.02203506976366043, "reward": 0.8100014328956604, "reward_std": 0.02203506790101528, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19199949502944946, "sampling/sampling_logp_difference/max": 1.4776219129562378, "sampling/importance_sampling_ratio/min": 0.2281796783208847, "sampling/importance_sampling_ratio/mean": 1.0370380878448486, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.443674087524414, "clip_ratio/low_mean": 0.04417553171515465, "clip_ratio/low_min": 0.04417553171515465, "clip_ratio/high_mean": 0.1387271974235773, "clip_ratio/high_max": 0.1387271974235773, "clip_ratio/region_mean": 0.18290272913873196, "reward_total_mean": 0.8100014328956604, "reward_meter_mean": 0.9825379252433777, "reward_meter_std": 0.03238074481487274, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9823436737060547, "reward_repeat_soft_std": 0.021364767104387283, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.8100014328956604, "reward_total_composite_std": 0.02203506976366043}