diff --git "a/train_stdout.log" "b/train_stdout.log" --- "a/train_stdout.log" +++ "b/train_stdout.log" @@ -737,3 +737,172 @@ The tokenizer has new PAD/BOS/EOS tokens that differ from the model config and g 6%|▌ | 201/3300 [28:15<19:47:27, 22.99s/it]2026-04-12 22:42:43,479 | INFO | train_grpo_train | metrics_logged mode=train step=201 {'loss': -0.0725, 'grad_norm': 9.45182991027832, 'learning_rate': 9.393939393939396e-06, 'num_tokens': 386624.0, 'completions/mean_length': 90.625, 'completions/min_length': 53.0, 'completions/max_length': 106.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 90.625, 'completions/min_terminated_length': 53.0, 'completions/max_terminated_length': 106.0, 'rewards/meter/mean': 0.7137892246246338, 'rewards/meter/std': 0.38705575466156006, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9941529035568237, 'rewards/repeat_soft/std': 0.004700822290033102, 'rewards/judge_quality/mean': 0.30124998092651367, 'rewards/judge_quality/std': 0.10398317128419876, 'rewards/total_composite/mean': 0.6609954833984375, 'rewards/total_composite/std': 0.19375428557395935, 'reward': 0.6609954833984375, 'reward_std': 0.19375428557395935, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.1688535064458847, 'sampling/sampling_logp_difference/max': 1.0816240310668945, 'sampling/importance_sampling_ratio/min': 0.339044451713562, 'sampling/importance_sampling_ratio/mean': 1.0338186025619507, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.1685265600681305, 'clip_ratio/low_mean': 0.04505446366965771, 'clip_ratio/low_min': 0.04505446366965771, 'clip_ratio/high_mean': 0.12919956259429455, 'clip_ratio/high_max': 0.12919956259429455, 'clip_ratio/region_mean': 0.17425402626395226, 'reward_total_mean': 0.6609954833984375, 'reward_meter_mean': 0.7137892246246338, 'reward_meter_std': 0.38705575466156006, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9941529035568237, 'reward_repeat_soft_std': 0.004700822290033102, 'reward_judge_quality_mean': 0.30124998092651367, 'reward_judge_quality_std': 0.10398317128419876, 'reward_total_composite_mean': 0.6609954833984375, 'reward_total_composite_std': 0.19375428557395935, 'epoch': 0.02} 6%|▌ | 201/3300 [28:15<19:47:27, 22.99s/it]INFO 04-12 22:42:43 [block_pool.py:378] Successfully reset prefix cache + 6%|▌ | 202/3300 [28:23<15:48:34, 18.37s/it]2026-04-12 22:42:51,074 | INFO | train_grpo_train | metrics_logged mode=train step=202 + {'loss': -0.0316, 'grad_norm': 8.221761703491211, 'learning_rate': 9.390909090909092e-06, 'num_tokens': 388803.0, 'completions/mean_length': 100.375, 'completions/min_length': 83.0, 'completions/max_length': 115.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 100.375, 'completions/min_terminated_length': 83.0, 'completions/max_terminated_length': 115.0, 'rewards/meter/mean': 0.9839462637901306, 'rewards/meter/std': 0.02068553864955902, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 0.75, 'rewards/hard_gate/std': 0.4629100561141968, 'rewards/repeat_soft/mean': 0.990378201007843, 'rewards/repeat_soft/std': 0.010706717148423195, 'rewards/judge_quality/mean': 0.32999998331069946, 'rewards/judge_quality/std': 0.11747339367866516, 'rewards/total_composite/mean': 0.5925955772399902, 'rewards/total_composite/std': 0.36694273352622986, 'reward': 0.5925955772399902, 'reward_std': 0.36694273352622986, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.1724347621202469, 'sampling/sampling_logp_difference/max': 1.2036924362182617, 'sampling/importance_sampling_ratio/min': 0.30008411407470703, 'sampling/importance_sampling_ratio/mean': 1.0368696451187134, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.376349061727524, 'clip_ratio/low_mean': 0.036620973609387875, 'clip_ratio/low_min': 0.036620973609387875, 'clip_ratio/high_mean': 0.10803817957639694, 'clip_ratio/high_max': 0.10803817957639694, 'clip_ratio/region_mean': 0.14465915318578482, 'reward_total_mean': 0.5925955772399902, 'reward_meter_mean': 0.9839462637901306, 'reward_meter_std': 0.02068553864955902, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 0.75, 'reward_hard_gate_std': 0.4629100561141968, 'reward_repeat_soft_mean': 0.990378201007843, 'reward_repeat_soft_std': 0.010706717148423195, 'reward_judge_quality_mean': 0.32999998331069946, 'reward_judge_quality_std': 0.11747339367866516, 'reward_total_composite_mean': 0.5925955772399902, 'reward_total_composite_std': 0.36694273352622986, 'epoch': 0.02} + 6%|▌ | 202/3300 [28:23<15:48:34, 18.37s/it]INFO 04-12 22:42:51 [block_pool.py:378] Successfully reset prefix cache + 6%|▌ | 203/3300 [28:32<13:19:37, 15.49s/it]2026-04-12 22:42:59,836 | INFO | train_grpo_train | metrics_logged mode=train step=203 + {'loss': -0.0311, 'grad_norm': 5.3910956382751465, 'learning_rate': 9.387878787878789e-06, 'num_tokens': 392045.0, 'completions/mean_length': 198.25, 'completions/min_length': 166.0, 'completions/max_length': 229.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 198.25, 'completions/min_terminated_length': 166.0, 'completions/max_terminated_length': 229.0, 'rewards/meter/mean': 0.8834912180900574, 'rewards/meter/std': 0.21112480759620667, 'rewards/count_adherence/mean': 0.875, 'rewards/count_adherence/std': 0.1035098284482956, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.992335855960846, 'rewards/repeat_soft/std': 0.006437603384256363, 'rewards/judge_quality/mean': 0.2800000011920929, 'rewards/judge_quality/std': 0.09304376691579819, 'rewards/total_composite/mean': 0.7120546102523804, 'rewards/total_composite/std': 0.07375628501176834, 'reward': 0.7120546102523804, 'reward_std': 0.07375627756118774, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.18845759332180023, 'sampling/sampling_logp_difference/max': 1.6991138458251953, 'sampling/importance_sampling_ratio/min': 0.18284548819065094, 'sampling/importance_sampling_ratio/mean': 1.0318385362625122, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.367601066827774, 'clip_ratio/low_mean': 0.05321737378835678, 'clip_ratio/low_min': 0.05321737378835678, 'clip_ratio/high_mean': 0.11454436741769314, 'clip_ratio/high_max': 0.11454436741769314, 'clip_ratio/region_mean': 0.16776174120604992, 'reward_total_mean': 0.7120546102523804, 'reward_meter_mean': 0.8834912180900574, 'reward_meter_std': 0.21112480759620667, 'reward_count_adherence_mean': 0.875, 'reward_count_adherence_std': 0.1035098284482956, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.992335855960846, 'reward_repeat_soft_std': 0.006437603384256363, 'reward_judge_quality_mean': 0.2800000011920929, 'reward_judge_quality_std': 0.09304376691579819, 'reward_total_composite_mean': 0.7120546102523804, 'reward_total_composite_std': 0.07375628501176834, 'epoch': 0.02} + 6%|▌ | 203/3300 [28:32<13:19:37, 15.49s/it]INFO 04-12 22:43:00 [block_pool.py:378] Successfully reset prefix cache + 6%|▌ | 204/3300 [28:39<11:04:29, 12.88s/it]2026-04-12 22:43:06,623 | INFO | train_grpo_train | metrics_logged mode=train step=204 + {'loss': 0.0376, 'grad_norm': 12.222681045532227, 'learning_rate': 9.384848484848486e-06, 'num_tokens': 393884.0, 'completions/mean_length': 58.875, 'completions/min_length': 44.0, 'completions/max_length': 66.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 58.875, 'completions/min_terminated_length': 44.0, 'completions/max_terminated_length': 66.0, 'rewards/meter/mean': 0.8527190089225769, 'rewards/meter/std': 0.23669300973415375, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 0.75, 'rewards/hard_gate/std': 0.4629100561141968, 'rewards/repeat_soft/mean': 0.9985201358795166, 'rewards/repeat_soft/std': 0.002760024508461356, 'rewards/judge_quality/mean': 0.3812499940395355, 'rewards/judge_quality/std': 0.08166787773370743, 'rewards/total_composite/mean': 0.5483859181404114, 'rewards/total_composite/std': 0.35711216926574707, 'reward': 0.5483859181404114, 'reward_std': 0.3571121394634247, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.19941037893295288, 'sampling/sampling_logp_difference/max': 1.3351926803588867, 'sampling/importance_sampling_ratio/min': 0.2631074786186218, 'sampling/importance_sampling_ratio/mean': 1.0467816591262817, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.5399347245693207, 'clip_ratio/low_mean': 0.061958421021699905, 'clip_ratio/low_min': 0.061958421021699905, 'clip_ratio/high_mean': 0.11895784921944141, 'clip_ratio/high_max': 0.11895784921944141, 'clip_ratio/region_mean': 0.18091627024114132, 'reward_total_mean': 0.5483859181404114, 'reward_meter_mean': 0.8527190089225769, 'reward_meter_std': 0.23669300973415375, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 0.75, 'reward_hard_gate_std': 0.4629100561141968, 'reward_repeat_soft_mean': 0.9985201358795166, 'reward_repeat_soft_std': 0.002760024508461356, 'reward_judge_quality_mean': 0.3812499940395355, 'reward_judge_quality_std': 0.08166787773370743, 'reward_total_composite_mean': 0.5483859181404114, 'reward_total_composite_std': 0.35711216926574707, 'epoch': 0.02} + 6%|▌ | 204/3300 [28:39<11:04:29, 12.88s/it]INFO 04-12 22:43:06 [block_pool.py:378] Successfully reset prefix cache + 6%|▌ | 205/3300 [28:46<9:44:23, 11.33s/it] 2026-04-12 22:43:14,344 | INFO | train_grpo_train | metrics_logged mode=train step=205 + {'loss': 0.0872, 'grad_norm': 6.21863317489624, 'learning_rate': 9.381818181818183e-06, 'num_tokens': 396523.0, 'completions/mean_length': 156.875, 'completions/min_length': 139.0, 'completions/max_length': 185.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 156.875, 'completions/min_terminated_length': 139.0, 'completions/max_terminated_length': 185.0, 'rewards/meter/mean': 0.9915645718574524, 'rewards/meter/std': 0.011944590136408806, 'rewards/count_adherence/mean': 0.90625, 'rewards/count_adherence/std': 0.12938730418682098, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9956042766571045, 'rewards/repeat_soft/std': 0.006392640061676502, 'rewards/judge_quality/mean': 0.3774999976158142, 'rewards/judge_quality/std': 0.07869470119476318, 'rewards/total_composite/mean': 0.7949520349502563, 'rewards/total_composite/std': 0.03068721294403076, 'reward': 0.7949520349502563, 'reward_std': 0.03068721294403076, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.1586761325597763, 'sampling/sampling_logp_difference/max': 1.254228115081787, 'sampling/importance_sampling_ratio/min': 0.2852959632873535, 'sampling/importance_sampling_ratio/mean': 1.029607892036438, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.133793383836746, 'clip_ratio/low_mean': 0.06302296090871096, 'clip_ratio/low_min': 0.06302296090871096, 'clip_ratio/high_mean': 0.0820816420018673, 'clip_ratio/high_max': 0.0820816420018673, 'clip_ratio/region_mean': 0.14510460291057825, 'reward_total_mean': 0.7949520349502563, 'reward_meter_mean': 0.9915645718574524, 'reward_meter_std': 0.011944590136408806, 'reward_count_adherence_mean': 0.90625, 'reward_count_adherence_std': 0.12938730418682098, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9956042766571045, 'reward_repeat_soft_std': 0.006392640061676502, 'reward_judge_quality_mean': 0.3774999976158142, 'reward_judge_quality_std': 0.07869470119476318, 'reward_total_composite_mean': 0.7949520349502563, 'reward_total_composite_std': 0.03068721294403076, 'epoch': 0.02} + 6%|▌ | 205/3300 [28:46<9:44:23, 11.33s/it]INFO 04-12 22:43:14 [block_pool.py:378] Successfully reset prefix cache + 6%|▌ | 206/3300 [28:53<8:33:04, 9.95s/it]2026-04-12 22:43:21,068 | INFO | train_grpo_train | metrics_logged mode=train step=206 + {'loss': 0.0365, 'grad_norm': 11.541095733642578, 'learning_rate': 9.378787878787879e-06, 'num_tokens': 398284.0, 'completions/mean_length': 60.125, 'completions/min_length': 43.0, 'completions/max_length': 71.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 60.125, 'completions/min_terminated_length': 43.0, 'completions/max_terminated_length': 71.0, 'rewards/meter/mean': 0.4893374443054199, 'rewards/meter/std': 0.3143778443336487, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9888759851455688, 'rewards/repeat_soft/std': 0.02070561796426773, 'rewards/judge_quality/mean': 0.4637500047683716, 'rewards/judge_quality/std': 0.2930596172809601, 'rewards/total_composite/mean': 0.6082144379615784, 'rewards/total_composite/std': 0.19916348159313202, 'reward': 0.6082144379615784, 'reward_std': 0.19916348159313202, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.20134250819683075, 'sampling/sampling_logp_difference/max': 1.501279354095459, 'sampling/importance_sampling_ratio/min': 0.22284488379955292, 'sampling/importance_sampling_ratio/mean': 1.034473180770874, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.1549899876117706, 'clip_ratio/low_mean': 0.12721418589353561, 'clip_ratio/low_min': 0.12721418589353561, 'clip_ratio/high_mean': 0.08245599828660488, 'clip_ratio/high_max': 0.08245599828660488, 'clip_ratio/region_mean': 0.2096701841801405, 'reward_total_mean': 0.6082144379615784, 'reward_meter_mean': 0.4893374443054199, 'reward_meter_std': 0.3143778443336487, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9888759851455688, 'reward_repeat_soft_std': 0.02070561796426773, 'reward_judge_quality_mean': 0.4637500047683716, 'reward_judge_quality_std': 0.2930596172809601, 'reward_total_composite_mean': 0.6082144379615784, 'reward_total_composite_std': 0.19916348159313202, 'epoch': 0.02} + 6%|▌ | 206/3300 [28:53<8:33:04, 9.95s/it]INFO 04-12 22:43:21 [block_pool.py:378] Successfully reset prefix cache + 6%|▋ | 207/3300 [29:00<7:44:21, 9.01s/it]2026-04-12 22:43:27,905 | INFO | train_grpo_train | metrics_logged mode=train step=207 + {'loss': 0.0986, 'grad_norm': 15.038732528686523, 'learning_rate': 9.375757575757576e-06, 'num_tokens': 400243.0, 'completions/mean_length': 75.875, 'completions/min_length': 57.0, 'completions/max_length': 94.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 75.875, 'completions/min_terminated_length': 57.0, 'completions/max_terminated_length': 94.0, 'rewards/meter/mean': 0.5208646059036255, 'rewards/meter/std': 0.3291780948638916, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9963343739509583, 'rewards/repeat_soft/std': 0.0034947823733091354, 'rewards/judge_quality/mean': 0.6700000166893005, 'rewards/judge_quality/std': 0.16903086006641388, 'rewards/total_composite/mean': 0.6850224733352661, 'rewards/total_composite/std': 0.17327746748924255, 'reward': 0.6850224733352661, 'reward_std': 0.17327745258808136, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.19959571957588196, 'sampling/sampling_logp_difference/max': 1.8032383918762207, 'sampling/importance_sampling_ratio/min': 0.16476444900035858, 'sampling/importance_sampling_ratio/mean': 1.0130186080932617, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.139797255396843, 'clip_ratio/low_mean': 0.1145754624158144, 'clip_ratio/low_min': 0.1145754624158144, 'clip_ratio/high_mean': 0.09555238112807274, 'clip_ratio/high_max': 0.09555238112807274, 'clip_ratio/region_mean': 0.21012784354388714, 'reward_total_mean': 0.6850224733352661, 'reward_meter_mean': 0.5208646059036255, 'reward_meter_std': 0.3291780948638916, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9963343739509583, 'reward_repeat_soft_std': 0.0034947823733091354, 'reward_judge_quality_mean': 0.6700000166893005, 'reward_judge_quality_std': 0.16903086006641388, 'reward_total_composite_mean': 0.6850224733352661, 'reward_total_composite_std': 0.17327746748924255, 'epoch': 0.02} + 6%|▋ | 207/3300 [29:00<7:44:21, 9.01s/it]INFO 04-12 22:43:28 [block_pool.py:378] Successfully reset prefix cache + 6%|▋ | 208/3300 [29:07<7:10:55, 8.36s/it]2026-04-12 22:43:34,766 | INFO | train_grpo_train | metrics_logged mode=train step=208 + {'loss': 0.0378, 'grad_norm': 16.253944396972656, 'learning_rate': 9.372727272727273e-06, 'num_tokens': 401689.0, 'completions/mean_length': 31.75, 'completions/min_length': 28.0, 'completions/max_length': 34.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 31.75, 'completions/min_terminated_length': 28.0, 'completions/max_terminated_length': 34.0, 'rewards/meter/mean': 0.6124137043952942, 'rewards/meter/std': 0.40621063113212585, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.938178300857544, 'rewards/repeat_soft/std': 0.027846481651067734, 'rewards/judge_quality/mean': 0.5275000333786011, 'rewards/judge_quality/std': 0.2499571591615677, 'rewards/total_composite/mean': 0.6776540279388428, 'rewards/total_composite/std': 0.19019806385040283, 'reward': 0.6776540279388428, 'reward_std': 0.19019807875156403, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.1393977701663971, 'sampling/sampling_logp_difference/max': 1.2818069458007812, 'sampling/importance_sampling_ratio/min': 0.2775353491306305, 'sampling/importance_sampling_ratio/mean': 1.0044059753417969, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.0049168691039085, 'clip_ratio/low_mean': 0.051893940195441246, 'clip_ratio/low_min': 0.051893940195441246, 'clip_ratio/high_mean': 0.07372742425650358, 'clip_ratio/high_max': 0.07372742425650358, 'clip_ratio/region_mean': 0.12562136445194483, 'reward_total_mean': 0.6776540279388428, 'reward_meter_mean': 0.6124137043952942, 'reward_meter_std': 0.40621063113212585, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.938178300857544, 'reward_repeat_soft_std': 0.027846481651067734, 'reward_judge_quality_mean': 0.5275000333786011, 'reward_judge_quality_std': 0.2499571591615677, 'reward_total_composite_mean': 0.6776540279388428, 'reward_total_composite_std': 0.19019806385040283, 'epoch': 0.02} + 6%|▋ | 208/3300 [29:07<7:10:55, 8.36s/it]INFO 04-12 22:43:35 [block_pool.py:378] Successfully reset prefix cache + 6%|▋ | 209/3300 [29:14<6:51:22, 7.99s/it]2026-04-12 22:43:41,853 | INFO | train_grpo_train | metrics_logged mode=train step=209 + {'loss': 0.0012, 'grad_norm': 10.391623497009277, 'learning_rate': 9.36969696969697e-06, 'num_tokens': 403420.0, 'completions/mean_length': 66.375, 'completions/min_length': 44.0, 'completions/max_length': 81.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 66.375, 'completions/min_terminated_length': 44.0, 'completions/max_terminated_length': 81.0, 'rewards/meter/mean': 0.6928176879882812, 'rewards/meter/std': 0.4111105501651764, 'rewards/count_adherence/mean': 0.9375, 'rewards/count_adherence/std': 0.1767766922712326, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9969114065170288, 'rewards/repeat_soft/std': 0.004841444082558155, 'rewards/judge_quality/mean': 0.4024999737739563, 'rewards/judge_quality/std': 0.16446885466575623, 'rewards/total_composite/mean': 0.6728341579437256, 'rewards/total_composite/std': 0.23347313702106476, 'reward': 0.6728341579437256, 'reward_std': 0.23347312211990356, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.17648614943027496, 'sampling/sampling_logp_difference/max': 1.2686376571655273, 'sampling/importance_sampling_ratio/min': 0.28121447563171387, 'sampling/importance_sampling_ratio/mean': 1.027706503868103, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.167846515774727, 'clip_ratio/low_mean': 0.02234848588705063, 'clip_ratio/low_min': 0.02234848588705063, 'clip_ratio/high_mean': 0.13506717514246702, 'clip_ratio/high_max': 0.13506717514246702, 'clip_ratio/region_mean': 0.15741566102951765, 'reward_total_mean': 0.6728341579437256, 'reward_meter_mean': 0.6928176879882812, 'reward_meter_std': 0.4111105501651764, 'reward_count_adherence_mean': 0.9375, 'reward_count_adherence_std': 0.1767766922712326, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9969114065170288, 'reward_repeat_soft_std': 0.004841444082558155, 'reward_judge_quality_mean': 0.4024999737739563, 'reward_judge_quality_std': 0.16446885466575623, 'reward_total_composite_mean': 0.6728341579437256, 'reward_total_composite_std': 0.23347313702106476, 'epoch': 0.02} + 6%|▋ | 209/3300 [29:14<6:51:22, 7.99s/it]INFO 04-12 22:43:42 [block_pool.py:378] Successfully reset prefix cache + 6%|▋ | 210/3300 [29:21<6:34:16, 7.66s/it]2026-04-12 22:43:48,738 | INFO | train_grpo_train | metrics_logged mode=train step=210 + {'loss': 0.0511, 'grad_norm': 7.208254337310791, 'learning_rate': 9.366666666666668e-06, 'num_tokens': 405820.0, 'completions/mean_length': 135.0, 'completions/min_length': 107.0, 'completions/max_length': 144.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 135.0, 'completions/min_terminated_length': 107.0, 'completions/max_terminated_length': 144.0, 'rewards/meter/mean': 0.9042514562606812, 'rewards/meter/std': 0.19750821590423584, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9914119243621826, 'rewards/repeat_soft/std': 0.0048458874225616455, 'rewards/judge_quality/mean': 0.3349999785423279, 'rewards/judge_quality/std': 0.09086881577968597, 'rewards/total_composite/mean': 0.756554365158081, 'rewards/total_composite/std': 0.08106998354196548, 'reward': 0.756554365158081, 'reward_std': 0.08106997609138489, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.16824375092983246, 'sampling/sampling_logp_difference/max': 1.6585102081298828, 'sampling/importance_sampling_ratio/min': 0.19042246043682098, 'sampling/importance_sampling_ratio/mean': 1.0354307889938354, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.138040006160736, 'clip_ratio/low_mean': 0.039393819868564606, 'clip_ratio/low_min': 0.039393819868564606, 'clip_ratio/high_mean': 0.13787824288010597, 'clip_ratio/high_max': 0.13787824288010597, 'clip_ratio/region_mean': 0.17727206274867058, 'reward_total_mean': 0.756554365158081, 'reward_meter_mean': 0.9042514562606812, 'reward_meter_std': 0.19750821590423584, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9914119243621826, 'reward_repeat_soft_std': 0.0048458874225616455, 'reward_judge_quality_mean': 0.3349999785423279, 'reward_judge_quality_std': 0.09086881577968597, 'reward_total_composite_mean': 0.756554365158081, 'reward_total_composite_std': 0.08106998354196548, 'epoch': 0.02} + 6%|▋ | 210/3300 [29:21<6:34:16, 7.66s/it]INFO 04-12 22:43:48 [block_pool.py:378] Successfully reset prefix cache + 6%|▋ | 211/3300 [29:27<6:08:08, 7.15s/it]2026-04-12 22:43:54,704 | INFO | train_grpo_train | metrics_logged mode=train step=211 + {'loss': 0.0704, 'grad_norm': 14.57507038116455, 'learning_rate': 9.363636363636365e-06, 'num_tokens': 407333.0, 'completions/mean_length': 29.125, 'completions/min_length': 17.0, 'completions/max_length': 33.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 29.125, 'completions/min_terminated_length': 17.0, 'completions/max_terminated_length': 33.0, 'rewards/meter/mean': 0.996258556842804, 'rewards/meter/std': 0.002240557223558426, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9624999761581421, 'rewards/repeat_soft/std': 0.0, 'rewards/judge_quality/mean': 0.34375, 'rewards/judge_quality/std': 0.10966669768095016, 'rewards/total_composite/mean': 0.7976913452148438, 'rewards/total_composite/std': 0.03349299728870392, 'reward': 0.7976913452148438, 'reward_std': 0.03349298983812332, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.1744379997253418, 'sampling/sampling_logp_difference/max': 1.1271677017211914, 'sampling/importance_sampling_ratio/min': 0.3239494860172272, 'sampling/importance_sampling_ratio/mean': 1.0234277248382568, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.7621784657239914, 'clip_ratio/low_mean': 0.10107611119747162, 'clip_ratio/low_min': 0.10107611119747162, 'clip_ratio/high_mean': 0.11261280253529549, 'clip_ratio/high_max': 0.11261280253529549, 'clip_ratio/region_mean': 0.2136889137327671, 'reward_total_mean': 0.7976913452148438, 'reward_meter_mean': 0.996258556842804, 'reward_meter_std': 0.002240557223558426, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9624999761581421, 'reward_repeat_soft_std': 0.0, 'reward_judge_quality_mean': 0.34375, 'reward_judge_quality_std': 0.10966669768095016, 'reward_total_composite_mean': 0.7976913452148438, 'reward_total_composite_std': 0.03349299728870392, 'epoch': 0.02} + 6%|▋ | 211/3300 [29:27<6:08:08, 7.15s/it]INFO 04-12 22:43:54 [block_pool.py:378] Successfully reset prefix cache + 6%|▋ | 212/3300 [29:34<6:10:11, 7.19s/it]2026-04-12 22:44:01,982 | INFO | train_grpo_train | metrics_logged mode=train step=212 + {'loss': 0.0435, 'grad_norm': 11.12042236328125, 'learning_rate': 9.36060606060606e-06, 'num_tokens': 409120.0, 'completions/mean_length': 65.375, 'completions/min_length': 60.0, 'completions/max_length': 71.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 65.375, 'completions/min_terminated_length': 60.0, 'completions/max_terminated_length': 71.0, 'rewards/meter/mean': 0.9937935471534729, 'rewards/meter/std': 0.008782095275819302, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9617397785186768, 'rewards/repeat_soft/std': 0.044201161712408066, 'rewards/judge_quality/mean': 0.4024999737739563, 'rewards/judge_quality/std': 0.06250713765621185, 'rewards/total_composite/mean': 0.8141311407089233, 'rewards/total_composite/std': 0.017935071140527725, 'reward': 0.8141311407089233, 'reward_std': 0.01793508231639862, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.1533195525407791, 'sampling/sampling_logp_difference/max': 1.0584931373596191, 'sampling/importance_sampling_ratio/min': 0.3782862722873688, 'sampling/importance_sampling_ratio/mean': 1.0483819246292114, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.9591315537691116, 'clip_ratio/low_mean': 0.027546045370399952, 'clip_ratio/low_min': 0.027546045370399952, 'clip_ratio/high_mean': 0.10376305785030127, 'clip_ratio/high_max': 0.10376305785030127, 'clip_ratio/region_mean': 0.13130910322070122, 'reward_total_mean': 0.8141311407089233, 'reward_meter_mean': 0.9937935471534729, 'reward_meter_std': 0.008782095275819302, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9617397785186768, 'reward_repeat_soft_std': 0.044201161712408066, 'reward_judge_quality_mean': 0.4024999737739563, 'reward_judge_quality_std': 0.06250713765621185, 'reward_total_composite_mean': 0.8141311407089233, 'reward_total_composite_std': 0.017935071140527725, 'epoch': 0.02} + 6%|▋ | 212/3300 [29:34<6:10:11, 7.19s/it]INFO 04-12 22:44:02 [block_pool.py:378] Successfully reset prefix cache + 6%|▋ | 213/3300 [29:42<6:17:59, 7.35s/it]2026-04-12 22:44:09,689 | INFO | train_grpo_train | metrics_logged mode=train step=213 + {'loss': 0.0364, 'grad_norm': 7.315183162689209, 'learning_rate': 9.357575757575758e-06, 'num_tokens': 411913.0, 'completions/mean_length': 165.125, 'completions/min_length': 145.0, 'completions/max_length': 189.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 165.125, 'completions/min_terminated_length': 145.0, 'completions/max_terminated_length': 189.0, 'rewards/meter/mean': 0.4947296977043152, 'rewards/meter/std': 0.29927948117256165, 'rewards/count_adherence/mean': 0.8999999761581421, 'rewards/count_adherence/std': 0.10690449178218842, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.996452808380127, 'rewards/repeat_soft/std': 0.0023691589012742043, 'rewards/judge_quality/mean': 0.35624998807907104, 'rewards/judge_quality/std': 0.08798335492610931, 'rewards/total_composite/mean': 0.5641486644744873, 'rewards/total_composite/std': 0.12801861763000488, 'reward': 0.5641486644744873, 'reward_std': 0.12801861763000488, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.1791200041770935, 'sampling/sampling_logp_difference/max': 1.5338325500488281, 'sampling/importance_sampling_ratio/min': 0.21570737659931183, 'sampling/importance_sampling_ratio/mean': 1.0269614458084106, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.175251916050911, 'clip_ratio/low_mean': 0.1019281204789877, 'clip_ratio/low_min': 0.1019281204789877, 'clip_ratio/high_mean': 0.07958623021841049, 'clip_ratio/high_max': 0.07958623021841049, 'clip_ratio/region_mean': 0.18151435069739819, 'reward_total_mean': 0.5641486644744873, 'reward_meter_mean': 0.4947296977043152, 'reward_meter_std': 0.29927948117256165, 'reward_count_adherence_mean': 0.8999999761581421, 'reward_count_adherence_std': 0.10690449178218842, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.996452808380127, 'reward_repeat_soft_std': 0.0023691589012742043, 'reward_judge_quality_mean': 0.35624998807907104, 'reward_judge_quality_std': 0.08798335492610931, 'reward_total_composite_mean': 0.5641486644744873, 'reward_total_composite_std': 0.12801861763000488, 'epoch': 0.02} + 6%|▋ | 213/3300 [29:42<6:17:59, 7.35s/it]INFO 04-12 22:44:09 [block_pool.py:378] Successfully reset prefix cache + 6%|▋ | 214/3300 [29:48<6:02:14, 7.04s/it]2026-04-12 22:44:16,023 | INFO | train_grpo_train | metrics_logged mode=train step=214 + {'loss': 0.053, 'grad_norm': 15.698607444763184, 'learning_rate': 9.354545454545455e-06, 'num_tokens': 413620.0, 'completions/mean_length': 54.375, 'completions/min_length': 46.0, 'completions/max_length': 64.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 54.375, 'completions/min_terminated_length': 46.0, 'completions/max_terminated_length': 64.0, 'rewards/meter/mean': 0.681374728679657, 'rewards/meter/std': 0.29896000027656555, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9897869825363159, 'rewards/repeat_soft/std': 0.01058239582926035, 'rewards/judge_quality/mean': 0.41874998807907104, 'rewards/judge_quality/std': 0.14574317634105682, 'rewards/total_composite/mean': 0.6812223196029663, 'rewards/total_composite/std': 0.12718971073627472, 'reward': 0.6812223196029663, 'reward_std': 0.1271897256374359, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.17061147093772888, 'sampling/sampling_logp_difference/max': 2.0034542083740234, 'sampling/importance_sampling_ratio/min': 0.13486862182617188, 'sampling/importance_sampling_ratio/mean': 1.0346450805664062, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.752282589673996, 'clip_ratio/low_mean': 0.07384132593870163, 'clip_ratio/low_min': 0.07384132593870163, 'clip_ratio/high_mean': 0.09478327073156834, 'clip_ratio/high_max': 0.09478327073156834, 'clip_ratio/region_mean': 0.16862459667026997, 'reward_total_mean': 0.6812223196029663, 'reward_meter_mean': 0.681374728679657, 'reward_meter_std': 0.29896000027656555, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9897869825363159, 'reward_repeat_soft_std': 0.01058239582926035, 'reward_judge_quality_mean': 0.41874998807907104, 'reward_judge_quality_std': 0.14574317634105682, 'reward_total_composite_mean': 0.6812223196029663, 'reward_total_composite_std': 0.12718971073627472, 'epoch': 0.02} + 6%|▋ | 214/3300 [29:48<6:02:14, 7.04s/it]INFO 04-12 22:44:16 [block_pool.py:378] Successfully reset prefix cache + 7%|▋ | 215/3300 [29:55<5:56:55, 6.94s/it]2026-04-12 22:44:22,730 | INFO | train_grpo_train | metrics_logged mode=train step=215 + {'loss': 0.0168, 'grad_norm': 9.017045021057129, 'learning_rate': 9.351515151515152e-06, 'num_tokens': 415817.0, 'completions/mean_length': 98.625, 'completions/min_length': 94.0, 'completions/max_length': 107.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 98.625, 'completions/min_terminated_length': 94.0, 'completions/max_terminated_length': 107.0, 'rewards/meter/mean': 0.7990944385528564, 'rewards/meter/std': 0.31399622559547424, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9968795776367188, 'rewards/repeat_soft/std': 0.0017273214180022478, 'rewards/judge_quality/mean': 0.5399999618530273, 'rewards/judge_quality/std': 0.22245386242866516, 'rewards/total_composite/mean': 0.7712804675102234, 'rewards/total_composite/std': 0.17402377724647522, 'reward': 0.7712804675102234, 'reward_std': 0.17402377724647522, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.1794540286064148, 'sampling/sampling_logp_difference/max': 2.4753665924072266, 'sampling/importance_sampling_ratio/min': 0.08413214236497879, 'sampling/importance_sampling_ratio/mean': 1.023207426071167, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.8376175910234451, 'clip_ratio/low_mean': 0.03789222054183483, 'clip_ratio/low_min': 0.03789222054183483, 'clip_ratio/high_mean': 0.09544784668833017, 'clip_ratio/high_max': 0.09544784668833017, 'clip_ratio/region_mean': 0.133340067230165, 'reward_total_mean': 0.7712804675102234, 'reward_meter_mean': 0.7990944385528564, 'reward_meter_std': 0.31399622559547424, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9968795776367188, 'reward_repeat_soft_std': 0.0017273214180022478, 'reward_judge_quality_mean': 0.5399999618530273, 'reward_judge_quality_std': 0.22245386242866516, 'reward_total_composite_mean': 0.7712804675102234, 'reward_total_composite_std': 0.17402377724647522, 'epoch': 0.02} + 7%|▋ | 215/3300 [29:55<5:56:55, 6.94s/it]INFO 04-12 22:44:22 [block_pool.py:378] Successfully reset prefix cache + 7%|▋ | 216/3300 [30:02<5:59:46, 7.00s/it]2026-04-12 22:44:29,863 | INFO | train_grpo_train | metrics_logged mode=train step=216 + {'loss': 0.111, 'grad_norm': 9.139263153076172, 'learning_rate': 9.34848484848485e-06, 'num_tokens': 418271.0, 'completions/mean_length': 92.75, 'completions/min_length': 76.0, 'completions/max_length': 115.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 92.75, 'completions/min_terminated_length': 76.0, 'completions/max_terminated_length': 115.0, 'rewards/meter/mean': 0.32860684394836426, 'rewards/meter/std': 0.3178398013114929, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9882336854934692, 'rewards/repeat_soft/std': 0.006412803195416927, 'rewards/judge_quality/mean': 0.38499999046325684, 'rewards/judge_quality/std': 0.23898595571517944, 'rewards/total_composite/mean': 0.51219642162323, 'rewards/total_composite/std': 0.19768090546131134, 'reward': 0.51219642162323, 'reward_std': 0.19768089056015015, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.20194599032402039, 'sampling/sampling_logp_difference/max': 2.3947439193725586, 'sampling/importance_sampling_ratio/min': 0.09119603037834167, 'sampling/importance_sampling_ratio/mean': 1.0125417709350586, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.4624213576316833, 'clip_ratio/low_mean': 0.13423257134854794, 'clip_ratio/low_min': 0.13423257134854794, 'clip_ratio/high_mean': 0.04835526458919048, 'clip_ratio/high_max': 0.04835526458919048, 'clip_ratio/region_mean': 0.18258783593773842, 'reward_total_mean': 0.51219642162323, 'reward_meter_mean': 0.32860684394836426, 'reward_meter_std': 0.3178398013114929, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9882336854934692, 'reward_repeat_soft_std': 0.006412803195416927, 'reward_judge_quality_mean': 0.38499999046325684, 'reward_judge_quality_std': 0.23898595571517944, 'reward_total_composite_mean': 0.51219642162323, 'reward_total_composite_std': 0.19768090546131134, 'epoch': 0.02} + 7%|▋ | 216/3300 [30:02<5:59:46, 7.00s/it]INFO 04-12 22:44:30 [block_pool.py:378] Successfully reset prefix cache + 7%|▋ | 217/3300 [30:10<6:23:18, 7.46s/it]2026-04-12 22:44:38,397 | INFO | train_grpo_train | metrics_logged mode=train step=217 + {'loss': 0.0786, 'grad_norm': 4.809051036834717, 'learning_rate': 9.345454545454547e-06, 'num_tokens': 421467.0, 'completions/mean_length': 200.5, 'completions/min_length': 170.0, 'completions/max_length': 217.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 200.5, 'completions/min_terminated_length': 170.0, 'completions/max_terminated_length': 217.0, 'rewards/meter/mean': 0.9879785776138306, 'rewards/meter/std': 0.020987501367926598, 'rewards/count_adherence/mean': 0.8999999761581421, 'rewards/count_adherence/std': 0.10690449178218842, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9745301008224487, 'rewards/repeat_soft/std': 0.019059404730796814, 'rewards/judge_quality/mean': 0.32249999046325684, 'rewards/judge_quality/std': 0.10925068706274033, 'rewards/total_composite/mean': 0.7737933397293091, 'rewards/total_composite/std': 0.0372769795358181, 'reward': 0.7737933397293091, 'reward_std': 0.037276968359947205, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.16808933019638062, 'sampling/sampling_logp_difference/max': 1.7900724411010742, 'sampling/importance_sampling_ratio/min': 0.16694806516170502, 'sampling/importance_sampling_ratio/mean': 1.037062406539917, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.176947444677353, 'clip_ratio/low_mean': 0.06853543221950531, 'clip_ratio/low_min': 0.06853543221950531, 'clip_ratio/high_mean': 0.06350381672382355, 'clip_ratio/high_max': 0.06350381672382355, 'clip_ratio/region_mean': 0.13203924894332886, 'reward_total_mean': 0.7737933397293091, 'reward_meter_mean': 0.9879785776138306, 'reward_meter_std': 0.020987501367926598, 'reward_count_adherence_mean': 0.8999999761581421, 'reward_count_adherence_std': 0.10690449178218842, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9745301008224487, 'reward_repeat_soft_std': 0.019059404730796814, 'reward_judge_quality_mean': 0.32249999046325684, 'reward_judge_quality_std': 0.10925068706274033, 'reward_total_composite_mean': 0.7737933397293091, 'reward_total_composite_std': 0.0372769795358181, 'epoch': 0.02} + 7%|▋ | 217/3300 [30:10<6:23:18, 7.46s/it]INFO 04-12 22:44:38 [block_pool.py:378] Successfully reset prefix cache + 7%|▋ | 218/3300 [30:17<6:06:40, 7.14s/it]2026-04-12 22:44:44,786 | INFO | train_grpo_train | metrics_logged mode=train step=218 + {'loss': 0.0107, 'grad_norm': 14.94848918914795, 'learning_rate': 9.342424242424243e-06, 'num_tokens': 423258.0, 'completions/mean_length': 55.875, 'completions/min_length': 36.0, 'completions/max_length': 66.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 55.875, 'completions/min_terminated_length': 36.0, 'completions/max_terminated_length': 66.0, 'rewards/meter/mean': 0.7060632705688477, 'rewards/meter/std': 0.38283491134643555, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9938966035842896, 'rewards/repeat_soft/std': 0.012834918685257435, 'rewards/judge_quality/mean': 0.42374998331069946, 'rewards/judge_quality/std': 0.010606602765619755, 'rewards/total_composite/mean': 0.6942431330680847, 'rewards/total_composite/std': 0.1732662171125412, 'reward': 0.6942431330680847, 'reward_std': 0.1732662469148636, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.2042960226535797, 'sampling/sampling_logp_difference/max': 1.2756810188293457, 'sampling/importance_sampling_ratio/min': 0.2792407274246216, 'sampling/importance_sampling_ratio/mean': 1.0386205911636353, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.241147205233574, 'clip_ratio/low_mean': 0.08530469797551632, 'clip_ratio/low_min': 0.08530469797551632, 'clip_ratio/high_mean': 0.12841526325792074, 'clip_ratio/high_max': 0.12841526325792074, 'clip_ratio/region_mean': 0.21371996123343706, 'reward_total_mean': 0.6942431330680847, 'reward_meter_mean': 0.7060632705688477, 'reward_meter_std': 0.38283491134643555, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9938966035842896, 'reward_repeat_soft_std': 0.012834918685257435, 'reward_judge_quality_mean': 0.42374998331069946, 'reward_judge_quality_std': 0.010606602765619755, 'reward_total_composite_mean': 0.6942431330680847, 'reward_total_composite_std': 0.1732662171125412, 'epoch': 0.02} + 7%|▋ | 218/3300 [30:17<6:06:40, 7.14s/it]INFO 04-12 22:44:45 [block_pool.py:378] Successfully reset prefix cache + 7%|▋ | 219/3300 [30:23<5:53:25, 6.88s/it]2026-04-12 22:44:51,072 | INFO | train_grpo_train | metrics_logged mode=train step=219 + {'loss': 0.0196, 'grad_norm': 11.719900131225586, 'learning_rate': 9.33939393939394e-06, 'num_tokens': 424945.0, 'completions/mean_length': 62.875, 'completions/min_length': 56.0, 'completions/max_length': 69.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 62.875, 'completions/min_terminated_length': 56.0, 'completions/max_terminated_length': 69.0, 'rewards/meter/mean': 0.6097301244735718, 'rewards/meter/std': 0.39728784561157227, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9982755780220032, 'rewards/repeat_soft/std': 0.00300014391541481, 'rewards/judge_quality/mean': 0.4612500071525574, 'rewards/judge_quality/std': 0.19467465579509735, 'rewards/total_composite/mean': 0.6625810861587524, 'rewards/total_composite/std': 0.21397148072719574, 'reward': 0.6625810861587524, 'reward_std': 0.21397146582603455, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.17527978122234344, 'sampling/sampling_logp_difference/max': 1.1357223987579346, 'sampling/importance_sampling_ratio/min': 0.3211899995803833, 'sampling/importance_sampling_ratio/mean': 1.044103980064392, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.2035064846277237, 'clip_ratio/low_mean': 0.041045167949050665, 'clip_ratio/low_min': 0.041045167949050665, 'clip_ratio/high_mean': 0.10810902435332537, 'clip_ratio/high_max': 0.10810902435332537, 'clip_ratio/region_mean': 0.14915419230237603, 'reward_total_mean': 0.6625810861587524, 'reward_meter_mean': 0.6097301244735718, 'reward_meter_std': 0.39728784561157227, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9982755780220032, 'reward_repeat_soft_std': 0.00300014391541481, 'reward_judge_quality_mean': 0.4612500071525574, 'reward_judge_quality_std': 0.19467465579509735, 'reward_total_composite_mean': 0.6625810861587524, 'reward_total_composite_std': 0.21397148072719574, 'epoch': 0.02} + 7%|▋ | 219/3300 [30:23<5:53:25, 6.88s/it]INFO 04-12 22:44:51 [block_pool.py:378] Successfully reset prefix cache + 7%|▋ | 220/3300 [30:29<5:42:32, 6.67s/it]2026-04-12 22:44:57,257 | INFO | train_grpo_train | metrics_logged mode=train step=220 + {'loss': 0.0964, 'grad_norm': 17.653593063354492, 'learning_rate': 9.336363636363637e-06, 'num_tokens': 426647.0, 'completions/mean_length': 47.75, 'completions/min_length': 39.0, 'completions/max_length': 57.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 47.75, 'completions/min_terminated_length': 39.0, 'completions/max_terminated_length': 57.0, 'rewards/meter/mean': 0.30497637391090393, 'rewards/meter/std': 0.34212055802345276, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.988140344619751, 'rewards/repeat_soft/std': 0.0252861138433218, 'rewards/judge_quality/mean': 0.6487500071525574, 'rewards/judge_quality/std': 0.2456151396036148, 'rewards/total_composite/mean': 0.580678403377533, 'rewards/total_composite/std': 0.18502292037010193, 'reward': 0.580678403377533, 'reward_std': 0.18502289056777954, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.19322755932807922, 'sampling/sampling_logp_difference/max': 2.5074644088745117, 'sampling/importance_sampling_ratio/min': 0.08147456496953964, 'sampling/importance_sampling_ratio/mean': 0.996791660785675, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.0714217349886894, 'clip_ratio/low_mean': 0.13291933946311474, 'clip_ratio/low_min': 0.13291933946311474, 'clip_ratio/high_mean': 0.07273874338716269, 'clip_ratio/high_max': 0.07273874338716269, 'clip_ratio/region_mean': 0.20565808285027742, 'reward_total_mean': 0.580678403377533, 'reward_meter_mean': 0.30497637391090393, 'reward_meter_std': 0.34212055802345276, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.988140344619751, 'reward_repeat_soft_std': 0.0252861138433218, 'reward_judge_quality_mean': 0.6487500071525574, 'reward_judge_quality_std': 0.2456151396036148, 'reward_total_composite_mean': 0.580678403377533, 'reward_total_composite_std': 0.18502292037010193, 'epoch': 0.02} + 7%|▋ | 220/3300 [30:29<5:42:32, 6.67s/it]INFO 04-12 22:44:57 [block_pool.py:378] Successfully reset prefix cache + 7%|▋ | 221/3300 [30:36<5:36:38, 6.56s/it]2026-04-12 22:45:03,552 | INFO | train_grpo_train | metrics_logged mode=train step=221 + {'loss': 0.1588, 'grad_norm': 18.759233474731445, 'learning_rate': 9.333333333333334e-06, 'num_tokens': 428076.0, 'completions/mean_length': 28.625, 'completions/min_length': 13.0, 'completions/max_length': 35.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 28.625, 'completions/min_terminated_length': 13.0, 'completions/max_terminated_length': 35.0, 'rewards/meter/mean': 0.6394675970077515, 'rewards/meter/std': 0.4421550929546356, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9472260475158691, 'rewards/repeat_soft/std': 0.028762657195329666, 'rewards/judge_quality/mean': 0.3787499964237213, 'rewards/judge_quality/std': 0.2436295449733734, 'rewards/total_composite/mean': 0.6461080312728882, 'rewards/total_composite/std': 0.22164414823055267, 'reward': 0.6461080312728882, 'reward_std': 0.22164413332939148, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.16225160658359528, 'sampling/sampling_logp_difference/max': 1.0092346668243408, 'sampling/importance_sampling_ratio/min': 0.3644978404045105, 'sampling/importance_sampling_ratio/mean': 1.0396852493286133, 'sampling/importance_sampling_ratio/max': 1.9327377080917358, 'entropy': 1.852670133113861, 'clip_ratio/low_mean': 0.030718903988599777, 'clip_ratio/low_min': 0.030718903988599777, 'clip_ratio/high_mean': 0.08502821158617735, 'clip_ratio/high_max': 0.08502821158617735, 'clip_ratio/region_mean': 0.11574711557477713, 'reward_total_mean': 0.6461080312728882, 'reward_meter_mean': 0.6394675970077515, 'reward_meter_std': 0.4421550929546356, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9472260475158691, 'reward_repeat_soft_std': 0.028762657195329666, 'reward_judge_quality_mean': 0.3787499964237213, 'reward_judge_quality_std': 0.2436295449733734, 'reward_total_composite_mean': 0.6461080312728882, 'reward_total_composite_std': 0.22164414823055267, 'epoch': 0.02} + 7%|▋ | 221/3300 [30:36<5:36:38, 6.56s/it]INFO 04-12 22:45:03 [block_pool.py:378] Successfully reset prefix cache + 7%|▋ | 222/3300 [30:42<5:28:44, 6.41s/it]2026-04-12 22:45:09,607 | INFO | train_grpo_train | metrics_logged mode=train step=222 + {'loss': 0.0564, 'grad_norm': 14.19981575012207, 'learning_rate': 9.33030303030303e-06, 'num_tokens': 429787.0, 'completions/mean_length': 50.875, 'completions/min_length': 38.0, 'completions/max_length': 57.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 50.875, 'completions/min_terminated_length': 38.0, 'completions/max_terminated_length': 57.0, 'rewards/meter/mean': 0.7463206052780151, 'rewards/meter/std': 0.35328948497772217, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9845018982887268, 'rewards/repeat_soft/std': 0.02211388200521469, 'rewards/judge_quality/mean': 0.40625, 'rewards/judge_quality/std': 0.0645727664232254, 'rewards/total_composite/mean': 0.7061694860458374, 'rewards/total_composite/std': 0.17570172250270844, 'reward': 0.7061694860458374, 'reward_std': 0.17570170760154724, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.17962686717510223, 'sampling/sampling_logp_difference/max': 1.4296016693115234, 'sampling/importance_sampling_ratio/min': 0.23940427601337433, 'sampling/importance_sampling_ratio/mean': 1.044725775718689, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.9504156559705734, 'clip_ratio/low_mean': 0.04155969247221947, 'clip_ratio/low_min': 0.04155969247221947, 'clip_ratio/high_mean': 0.12596898153424263, 'clip_ratio/high_max': 0.12596898153424263, 'clip_ratio/region_mean': 0.1675286740064621, 'reward_total_mean': 0.7061694860458374, 'reward_meter_mean': 0.7463206052780151, 'reward_meter_std': 0.35328948497772217, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9845018982887268, 'reward_repeat_soft_std': 0.02211388200521469, 'reward_judge_quality_mean': 0.40625, 'reward_judge_quality_std': 0.0645727664232254, 'reward_total_composite_mean': 0.7061694860458374, 'reward_total_composite_std': 0.17570172250270844, 'epoch': 0.02} + 7%|▋ | 222/3300 [30:42<5:28:44, 6.41s/it]INFO 04-12 22:45:09 [block_pool.py:378] Successfully reset prefix cache + 7%|▋ | 223/3300 [30:48<5:34:56, 6.53s/it]2026-04-12 22:45:16,424 | INFO | train_grpo_train | metrics_logged mode=train step=223 + {'loss': 0.0257, 'grad_norm': 7.719682693481445, 'learning_rate': 9.327272727272729e-06, 'num_tokens': 431970.0, 'completions/mean_length': 100.875, 'completions/min_length': 89.0, 'completions/max_length': 118.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 100.875, 'completions/min_terminated_length': 89.0, 'completions/max_terminated_length': 118.0, 'rewards/meter/mean': 0.9940782785415649, 'rewards/meter/std': 0.002984443213790655, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 0.875, 'rewards/hard_gate/std': 0.3535533845424652, 'rewards/repeat_soft/mean': 0.9923332929611206, 'rewards/repeat_soft/std': 0.007202796172350645, 'rewards/judge_quality/mean': 0.3349999785423279, 'rewards/judge_quality/std': 0.09086881577968597, 'rewards/total_composite/mean': 0.7004891633987427, 'rewards/total_composite/std': 0.2841547727584839, 'reward': 0.7004891633987427, 'reward_std': 0.2841547429561615, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.18082353472709656, 'sampling/sampling_logp_difference/max': 1.2701425552368164, 'sampling/importance_sampling_ratio/min': 0.28079158067703247, 'sampling/importance_sampling_ratio/mean': 1.0449841022491455, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.545370638370514, 'clip_ratio/low_mean': 0.015776699408888817, 'clip_ratio/low_min': 0.015776699408888817, 'clip_ratio/high_mean': 0.1305677006021142, 'clip_ratio/high_max': 0.1305677006021142, 'clip_ratio/region_mean': 0.14634440001100302, 'reward_total_mean': 0.7004891633987427, 'reward_meter_mean': 0.9940782785415649, 'reward_meter_std': 0.002984443213790655, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 0.875, 'reward_hard_gate_std': 0.3535533845424652, 'reward_repeat_soft_mean': 0.9923332929611206, 'reward_repeat_soft_std': 0.007202796172350645, 'reward_judge_quality_mean': 0.3349999785423279, 'reward_judge_quality_std': 0.09086881577968597, 'reward_total_composite_mean': 0.7004891633987427, 'reward_total_composite_std': 0.2841547727584839, 'epoch': 0.02} + 7%|▋ | 223/3300 [30:48<5:34:56, 6.53s/it]INFO 04-12 22:45:16 [block_pool.py:378] Successfully reset prefix cache + 7%|▋ | 224/3300 [30:55<5:35:15, 6.54s/it]2026-04-12 22:45:22,987 | INFO | train_grpo_train | metrics_logged mode=train step=224 + {'loss': -0.0319, 'grad_norm': 9.542694091796875, 'learning_rate': 9.324242424242424e-06, 'num_tokens': 433859.0, 'completions/mean_length': 83.125, 'completions/min_length': 63.0, 'completions/max_length': 92.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 83.125, 'completions/min_terminated_length': 63.0, 'completions/max_terminated_length': 92.0, 'rewards/meter/mean': 0.954963207244873, 'rewards/meter/std': 0.05485004186630249, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9916623830795288, 'rewards/repeat_soft/std': 0.004886186216026545, 'rewards/judge_quality/mean': 0.4475000202655792, 'rewards/judge_quality/std': 0.12848013639450073, 'rewards/total_composite/mean': 0.8131496906280518, 'rewards/total_composite/std': 0.047446589916944504, 'reward': 0.8131496906280518, 'reward_std': 0.0474465973675251, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.1831963062286377, 'sampling/sampling_logp_difference/max': 1.5137519836425781, 'sampling/importance_sampling_ratio/min': 0.22008268535137177, 'sampling/importance_sampling_ratio/mean': 1.0295239686965942, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.975585401058197, 'clip_ratio/low_mean': 0.05584507342427969, 'clip_ratio/low_min': 0.05584507342427969, 'clip_ratio/high_mean': 0.11094464547932148, 'clip_ratio/high_max': 0.11094464547932148, 'clip_ratio/region_mean': 0.16678971890360117, 'reward_total_mean': 0.8131496906280518, 'reward_meter_mean': 0.954963207244873, 'reward_meter_std': 0.05485004186630249, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9916623830795288, 'reward_repeat_soft_std': 0.004886186216026545, 'reward_judge_quality_mean': 0.4475000202655792, 'reward_judge_quality_std': 0.12848013639450073, 'reward_total_composite_mean': 0.8131496906280518, 'reward_total_composite_std': 0.047446589916944504, 'epoch': 0.02} + 7%|▋ | 224/3300 [30:55<5:35:15, 6.54s/it]INFO 04-12 22:45:23 [block_pool.py:378] Successfully reset prefix cache + 7%|▋ | 225/3300 [31:03<5:51:58, 6.87s/it]2026-04-12 22:45:30,617 | INFO | train_grpo_train | metrics_logged mode=train step=225 + {'loss': 0.0604, 'grad_norm': 9.695013046264648, 'learning_rate': 9.321212121212122e-06, 'num_tokens': 436413.0, 'completions/mean_length': 119.25, 'completions/min_length': 94.0, 'completions/max_length': 146.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 119.25, 'completions/min_terminated_length': 94.0, 'completions/max_terminated_length': 146.0, 'rewards/meter/mean': 0.5250018835067749, 'rewards/meter/std': 0.3770351707935333, 'rewards/count_adherence/mean': 0.96875, 'rewards/count_adherence/std': 0.0883883461356163, 'rewards/hard_gate/mean': 0.875, 'rewards/hard_gate/std': 0.3535533845424652, 'rewards/repeat_soft/mean': 0.9959819912910461, 'rewards/repeat_soft/std': 0.0034585981629788876, 'rewards/judge_quality/mean': 0.32249999046325684, 'rewards/judge_quality/std': 0.10925068706274033, 'rewards/total_composite/mean': 0.4833066463470459, 'rewards/total_composite/std': 0.2536589801311493, 'reward': 0.4833066463470459, 'reward_std': 0.2536589801311493, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.19236743450164795, 'sampling/sampling_logp_difference/max': 1.2519326210021973, 'sampling/importance_sampling_ratio/min': 0.2859516441822052, 'sampling/importance_sampling_ratio/mean': 1.0415294170379639, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.555527776479721, 'clip_ratio/low_mean': 0.06408403068780899, 'clip_ratio/low_min': 0.06408403068780899, 'clip_ratio/high_mean': 0.12092001549899578, 'clip_ratio/high_max': 0.12092001549899578, 'clip_ratio/region_mean': 0.18500404618680477, 'reward_total_mean': 0.4833066463470459, 'reward_meter_mean': 0.5250018835067749, 'reward_meter_std': 0.3770351707935333, 'reward_count_adherence_mean': 0.96875, 'reward_count_adherence_std': 0.0883883461356163, 'reward_hard_gate_mean': 0.875, 'reward_hard_gate_std': 0.3535533845424652, 'reward_repeat_soft_mean': 0.9959819912910461, 'reward_repeat_soft_std': 0.0034585981629788876, 'reward_judge_quality_mean': 0.32249999046325684, 'reward_judge_quality_std': 0.10925068706274033, 'reward_total_composite_mean': 0.4833066463470459, 'reward_total_composite_std': 0.2536589801311493, 'epoch': 0.02} + 7%|▋ | 225/3300 [31:03<5:51:58, 6.87s/it]INFO 04-12 22:45:30 [block_pool.py:378] Successfully reset prefix cache + 7%|▋ | 226/3300 [31:08<5:33:56, 6.52s/it]2026-04-12 22:45:36,320 | INFO | train_grpo_train | metrics_logged mode=train step=226 + {'loss': -0.0401, 'grad_norm': 13.432271003723145, 'learning_rate': 9.318181818181819e-06, 'num_tokens': 437832.0, 'completions/mean_length': 27.375, 'completions/min_length': 22.0, 'completions/max_length': 31.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 27.375, 'completions/min_terminated_length': 22.0, 'completions/max_terminated_length': 31.0, 'rewards/meter/mean': 0.7391682863235474, 'rewards/meter/std': 0.24732904136180878, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9533430337905884, 'rewards/repeat_soft/std': 0.0222688727080822, 'rewards/judge_quality/mean': 0.4387499988079071, 'rewards/judge_quality/std': 0.015526476316154003, 'rewards/total_composite/mean': 0.7095850110054016, 'rewards/total_composite/std': 0.11227390915155411, 'reward': 0.7095850110054016, 'reward_std': 0.11227390915155411, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.1079363226890564, 'sampling/sampling_logp_difference/max': 0.9618256092071533, 'sampling/importance_sampling_ratio/min': 0.38219451904296875, 'sampling/importance_sampling_ratio/mean': 1.016780138015747, 'sampling/importance_sampling_ratio/max': 1.5922904014587402, 'entropy': 0.8746723867952824, 'clip_ratio/low_mean': 0.08646365813910961, 'clip_ratio/low_min': 0.08646365813910961, 'clip_ratio/high_mean': 0.07067448738962412, 'clip_ratio/high_max': 0.07067448738962412, 'clip_ratio/region_mean': 0.15713814552873373, 'reward_total_mean': 0.7095850110054016, 'reward_meter_mean': 0.7391682863235474, 'reward_meter_std': 0.24732904136180878, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9533430337905884, 'reward_repeat_soft_std': 0.0222688727080822, 'reward_judge_quality_mean': 0.4387499988079071, 'reward_judge_quality_std': 0.015526476316154003, 'reward_total_composite_mean': 0.7095850110054016, 'reward_total_composite_std': 0.11227390915155411, 'epoch': 0.02} + 7%|▋ | 226/3300 [31:08<5:33:56, 6.52s/it]INFO 04-12 22:45:36 [block_pool.py:378] Successfully reset prefix cache + 7%|▋ | 227/3300 [31:16<5:49:44, 6.83s/it]2026-04-12 22:45:43,872 | INFO | train_grpo_train | metrics_logged mode=train step=227 + {'loss': 0.0181, 'grad_norm': 6.291835308074951, 'learning_rate': 9.315151515151516e-06, 'num_tokens': 440362.0, 'completions/mean_length': 140.25, 'completions/min_length': 132.0, 'completions/max_length': 151.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 140.25, 'completions/min_terminated_length': 132.0, 'completions/max_terminated_length': 151.0, 'rewards/meter/mean': 0.8960759043693542, 'rewards/meter/std': 0.20459839701652527, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.996003270149231, 'rewards/repeat_soft/std': 0.004744499456137419, 'rewards/judge_quality/mean': 0.3349999785423279, 'rewards/judge_quality/std': 0.09086881577968597, 'rewards/total_composite/mean': 0.7533344626426697, 'rewards/total_composite/std': 0.08933799713850021, 'reward': 0.7533344626426697, 'reward_std': 0.08933799713850021, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.1747959405183792, 'sampling/sampling_logp_difference/max': 2.201197624206543, 'sampling/importance_sampling_ratio/min': 0.1106705367565155, 'sampling/importance_sampling_ratio/mean': 1.0493431091308594, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.3742400407791138, 'clip_ratio/low_mean': 0.037852587178349495, 'clip_ratio/low_min': 0.037852587178349495, 'clip_ratio/high_mean': 0.11660830955952406, 'clip_ratio/high_max': 0.11660830955952406, 'clip_ratio/region_mean': 0.15446089673787355, 'reward_total_mean': 0.7533344626426697, 'reward_meter_mean': 0.8960759043693542, 'reward_meter_std': 0.20459839701652527, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.996003270149231, 'reward_repeat_soft_std': 0.004744499456137419, 'reward_judge_quality_mean': 0.3349999785423279, 'reward_judge_quality_std': 0.09086881577968597, 'reward_total_composite_mean': 0.7533344626426697, 'reward_total_composite_std': 0.08933799713850021, 'epoch': 0.02} + 7%|▋ | 227/3300 [31:16<5:49:44, 6.83s/it]INFO 04-12 22:45:44 [block_pool.py:378] Successfully reset prefix cache + 7%|▋ | 228/3300 [31:22<5:43:26, 6.71s/it]2026-04-12 22:45:50,300 | INFO | train_grpo_train | metrics_logged mode=train step=228 + {'loss': 0.024, 'grad_norm': 10.584402084350586, 'learning_rate': 9.312121212121212e-06, 'num_tokens': 442195.0, 'completions/mean_length': 59.125, 'completions/min_length': 54.0, 'completions/max_length': 64.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 59.125, 'completions/min_terminated_length': 54.0, 'completions/max_terminated_length': 64.0, 'rewards/meter/mean': 0.9284141063690186, 'rewards/meter/std': 0.11800399422645569, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9977113008499146, 'rewards/repeat_soft/std': 0.0055434200912714005, 'rewards/judge_quality/mean': 0.48500001430511475, 'rewards/judge_quality/std': 0.2285982221364975, 'rewards/total_composite/mean': 0.813057541847229, 'rewards/total_composite/std': 0.0972953587770462, 'reward': 0.813057541847229, 'reward_std': 0.0972953587770462, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.17132289707660675, 'sampling/sampling_logp_difference/max': 1.5689716339111328, 'sampling/importance_sampling_ratio/min': 0.2082592397928238, 'sampling/importance_sampling_ratio/mean': 1.0282717943191528, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.9079337120056152, 'clip_ratio/low_mean': 0.08830753713846207, 'clip_ratio/low_min': 0.08830753713846207, 'clip_ratio/high_mean': 0.08729508332908154, 'clip_ratio/high_max': 0.08729508332908154, 'clip_ratio/region_mean': 0.1756026204675436, 'reward_total_mean': 0.813057541847229, 'reward_meter_mean': 0.9284141063690186, 'reward_meter_std': 0.11800399422645569, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9977113008499146, 'reward_repeat_soft_std': 0.0055434200912714005, 'reward_judge_quality_mean': 0.48500001430511475, 'reward_judge_quality_std': 0.2285982221364975, 'reward_total_composite_mean': 0.813057541847229, 'reward_total_composite_std': 0.0972953587770462, 'epoch': 0.02} + 7%|▋ | 228/3300 [31:22<5:43:26, 6.71s/it]INFO 04-12 22:45:50 [block_pool.py:378] Successfully reset prefix cache + 7%|▋ | 229/3300 [31:29<5:39:55, 6.64s/it]2026-04-12 22:45:56,785 | INFO | train_grpo_train | metrics_logged mode=train step=229 + {'loss': -0.0087, 'grad_norm': 11.501396179199219, 'learning_rate': 9.30909090909091e-06, 'num_tokens': 443906.0, 'completions/mean_length': 57.875, 'completions/min_length': 35.0, 'completions/max_length': 66.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 57.875, 'completions/min_terminated_length': 35.0, 'completions/max_terminated_length': 66.0, 'rewards/meter/mean': 0.978620171546936, 'rewards/meter/std': 0.019945673644542694, 'rewards/count_adherence/mean': 0.9375, 'rewards/count_adherence/std': 0.1767766922712326, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9931089878082275, 'rewards/repeat_soft/std': 0.011518539860844612, 'rewards/judge_quality/mean': 0.40625, 'rewards/judge_quality/std': 0.0645727664232254, 'rewards/total_composite/mean': 0.8021900057792664, 'rewards/total_composite/std': 0.03168172389268875, 'reward': 0.8021900057792664, 'reward_std': 0.03168172389268875, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.16884812712669373, 'sampling/sampling_logp_difference/max': 1.5398664474487305, 'sampling/importance_sampling_ratio/min': 0.214409738779068, 'sampling/importance_sampling_ratio/mean': 1.0520275831222534, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.3012482821941376, 'clip_ratio/low_mean': 0.07348901219666004, 'clip_ratio/low_min': 0.07348901219666004, 'clip_ratio/high_mean': 0.12493743561208248, 'clip_ratio/high_max': 0.12493743561208248, 'clip_ratio/region_mean': 0.19842644780874252, 'reward_total_mean': 0.8021900057792664, 'reward_meter_mean': 0.978620171546936, 'reward_meter_std': 0.019945673644542694, 'reward_count_adherence_mean': 0.9375, 'reward_count_adherence_std': 0.1767766922712326, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9931089878082275, 'reward_repeat_soft_std': 0.011518539860844612, 'reward_judge_quality_mean': 0.40625, 'reward_judge_quality_std': 0.0645727664232254, 'reward_total_composite_mean': 0.8021900057792664, 'reward_total_composite_std': 0.03168172389268875, 'epoch': 0.02} + 7%|▋ | 229/3300 [31:29<5:39:55, 6.64s/it]INFO 04-12 22:45:57 [block_pool.py:378] Successfully reset prefix cache + 7%|▋ | 230/3300 [31:35<5:40:43, 6.66s/it]2026-04-12 22:46:03,938 | INFO | train_grpo_train | metrics_logged mode=train step=230 + {'loss': 0.1247, 'grad_norm': 15.090785026550293, 'learning_rate': 9.306060606060608e-06, 'num_tokens': 445916.0, 'completions/mean_length': 67.25, 'completions/min_length': 62.0, 'completions/max_length': 86.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 67.25, 'completions/min_terminated_length': 62.0, 'completions/max_terminated_length': 86.0, 'rewards/meter/mean': 0.6055946350097656, 'rewards/meter/std': 0.43195900321006775, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.982811450958252, 'rewards/repeat_soft/std': 0.016159115359187126, 'rewards/judge_quality/mean': 0.3987500071525574, 'rewards/judge_quality/std': 0.06010407209396362, 'rewards/total_composite/mean': 0.6404237747192383, 'rewards/total_composite/std': 0.20140200853347778, 'reward': 0.6404237747192383, 'reward_std': 0.2014019936323166, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.19653742015361786, 'sampling/sampling_logp_difference/max': 2.1093668937683105, 'sampling/importance_sampling_ratio/min': 0.12131474912166595, 'sampling/importance_sampling_ratio/mean': 0.9851779937744141, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.5822003185749054, 'clip_ratio/low_mean': 0.0612277602776885, 'clip_ratio/low_min': 0.0612277602776885, 'clip_ratio/high_mean': 0.11418783850967884, 'clip_ratio/high_max': 0.11418783850967884, 'clip_ratio/region_mean': 0.17541559878736734, 'reward_total_mean': 0.6404237747192383, 'reward_meter_mean': 0.6055946350097656, 'reward_meter_std': 0.43195900321006775, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.982811450958252, 'reward_repeat_soft_std': 0.016159115359187126, 'reward_judge_quality_mean': 0.3987500071525574, 'reward_judge_quality_std': 0.06010407209396362, 'reward_total_composite_mean': 0.6404237747192383, 'reward_total_composite_std': 0.20140200853347778, 'epoch': 0.02} + 7%|▋ | 230/3300 [31:36<5:40:43, 6.66s/it]INFO 04-12 22:46:04 [block_pool.py:378] Successfully reset prefix cache + 7%|▋ | 231/3300 [31:43<5:54:05, 6.92s/it]2026-04-12 22:46:11,023 | INFO | train_grpo_train | metrics_logged mode=train step=231 + {'loss': 0.0272, 'grad_norm': 9.040948867797852, 'learning_rate': 9.303030303030303e-06, 'num_tokens': 447745.0, 'completions/mean_length': 64.625, 'completions/min_length': 49.0, 'completions/max_length': 69.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 64.625, 'completions/min_terminated_length': 49.0, 'completions/max_terminated_length': 69.0, 'rewards/meter/mean': 0.989717960357666, 'rewards/meter/std': 0.01920364797115326, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9972941875457764, 'rewards/repeat_soft/std': 0.004934768658131361, 'rewards/judge_quality/mean': 0.5024999976158142, 'rewards/judge_quality/std': 0.16705432534217834, 'rewards/total_composite/mean': 0.8458524942398071, 'rewards/total_composite/std': 0.052879635244607925, 'reward': 0.8458524942398071, 'reward_std': 0.05287962406873703, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.15886950492858887, 'sampling/sampling_logp_difference/max': 1.3513360023498535, 'sampling/importance_sampling_ratio/min': 0.258894145488739, 'sampling/importance_sampling_ratio/mean': 1.0313369035720825, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.0065972805023193, 'clip_ratio/low_mean': 0.08278274443000555, 'clip_ratio/low_min': 0.08278274443000555, 'clip_ratio/high_mean': 0.054113051854074, 'clip_ratio/high_max': 0.054113051854074, 'clip_ratio/region_mean': 0.13689579628407955, 'reward_total_mean': 0.8458524942398071, 'reward_meter_mean': 0.989717960357666, 'reward_meter_std': 0.01920364797115326, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9972941875457764, 'reward_repeat_soft_std': 0.004934768658131361, 'reward_judge_quality_mean': 0.5024999976158142, 'reward_judge_quality_std': 0.16705432534217834, 'reward_total_composite_mean': 0.8458524942398071, 'reward_total_composite_std': 0.052879635244607925, 'epoch': 0.02} + 7%|▋ | 231/3300 [31:43<5:54:05, 6.92s/it]INFO 04-12 22:46:11 [block_pool.py:378] Successfully reset prefix cache + 7%|▋ | 232/3300 [31:50<5:50:41, 6.86s/it]2026-04-12 22:46:18,082 | INFO | train_grpo_train | metrics_logged mode=train step=232 + {'loss': 0.0074, 'grad_norm': 9.7235107421875, 'learning_rate': 9.3e-06, 'num_tokens': 449644.0, 'completions/mean_length': 71.375, 'completions/min_length': 67.0, 'completions/max_length': 78.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 71.375, 'completions/min_terminated_length': 67.0, 'completions/max_terminated_length': 78.0, 'rewards/meter/mean': 0.8106899261474609, 'rewards/meter/std': 0.32043975591659546, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9995647668838501, 'rewards/repeat_soft/std': 0.0011147403856739402, 'rewards/judge_quality/mean': 0.7950000166893005, 'rewards/judge_quality/std': 0.23145504295825958, 'rewards/total_composite/mean': 0.8532669544219971, 'rewards/total_composite/std': 0.1411026418209076, 'reward': 0.8532669544219971, 'reward_std': 0.1411026567220688, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.16306047141551971, 'sampling/sampling_logp_difference/max': 1.0975570678710938, 'sampling/importance_sampling_ratio/min': 0.3336852490901947, 'sampling/importance_sampling_ratio/mean': 1.0272496938705444, 'sampling/importance_sampling_ratio/max': 1.9878556728363037, 'entropy': 2.0106695145368576, 'clip_ratio/low_mean': 0.05744114611297846, 'clip_ratio/low_min': 0.05744114611297846, 'clip_ratio/high_mean': 0.07862667366862297, 'clip_ratio/high_max': 0.07862667366862297, 'clip_ratio/region_mean': 0.13606781978160143, 'reward_total_mean': 0.8532669544219971, 'reward_meter_mean': 0.8106899261474609, 'reward_meter_std': 0.32043975591659546, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9995647668838501, 'reward_repeat_soft_std': 0.0011147403856739402, 'reward_judge_quality_mean': 0.7950000166893005, 'reward_judge_quality_std': 0.23145504295825958, 'reward_total_composite_mean': 0.8532669544219971, 'reward_total_composite_std': 0.1411026418209076, 'epoch': 0.02} + 7%|▋ | 232/3300 [31:50<5:50:41, 6.86s/it]INFO 04-12 22:46:18 [block_pool.py:378] Successfully reset prefix cache + 7%|▋ | 233/3300 [32:00<6:44:42, 7.92s/it]2026-04-12 22:46:28,124 | INFO | train_grpo_train | metrics_logged mode=train step=233 + {'loss': 0.0133, 'grad_norm': 6.52683687210083, 'learning_rate': 9.296969696969698e-06, 'num_tokens': 452785.0, 'completions/mean_length': 200.625, 'completions/min_length': 155.0, 'completions/max_length': 242.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 200.625, 'completions/min_terminated_length': 155.0, 'completions/max_terminated_length': 242.0, 'rewards/meter/mean': 0.35411587357521057, 'rewards/meter/std': 0.3424142599105835, 'rewards/count_adherence/mean': 0.8333333134651184, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9926233291625977, 'rewards/repeat_soft/std': 0.006160663906484842, 'rewards/judge_quality/mean': 0.2887499928474426, 'rewards/judge_quality/std': 0.11630470305681229, 'rewards/total_composite/mean': 0.4702394902706146, 'rewards/total_composite/std': 0.16173307597637177, 'reward': 0.4702394902706146, 'reward_std': 0.16173309087753296, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.19548551738262177, 'sampling/sampling_logp_difference/max': 1.9297070503234863, 'sampling/importance_sampling_ratio/min': 0.14519073069095612, 'sampling/importance_sampling_ratio/mean': 1.037513256072998, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.861027091741562, 'clip_ratio/low_mean': 0.09871664829552174, 'clip_ratio/low_min': 0.09871664829552174, 'clip_ratio/high_mean': 0.08370550721883774, 'clip_ratio/high_max': 0.08370550721883774, 'clip_ratio/region_mean': 0.18242215551435947, 'reward_total_mean': 0.4702394902706146, 'reward_meter_mean': 0.35411587357521057, 'reward_meter_std': 0.3424142599105835, 'reward_count_adherence_mean': 0.8333333134651184, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9926233291625977, 'reward_repeat_soft_std': 0.006160663906484842, 'reward_judge_quality_mean': 0.2887499928474426, 'reward_judge_quality_std': 0.11630470305681229, 'reward_total_composite_mean': 0.4702394902706146, 'reward_total_composite_std': 0.16173307597637177, 'epoch': 0.02} + 7%|▋ | 233/3300 [32:00<6:44:42, 7.92s/it]INFO 04-12 22:46:28 [block_pool.py:378] Successfully reset prefix cache + 7%|▋ | 234/3300 [32:11<7:26:51, 8.74s/it]2026-04-12 22:46:38,802 | INFO | train_grpo_train | metrics_logged mode=train step=234 + {'loss': 0.0747, 'grad_norm': 12.53782844543457, 'learning_rate': 9.293939393939395e-06, 'num_tokens': 454263.0, 'completions/mean_length': 28.75, 'completions/min_length': 23.0, 'completions/max_length': 34.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 28.75, 'completions/min_terminated_length': 23.0, 'completions/max_terminated_length': 34.0, 'rewards/meter/mean': 0.6159642934799194, 'rewards/meter/std': 0.4214301109313965, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9624999761581421, 'rewards/repeat_soft/std': 0.0, 'rewards/judge_quality/mean': 0.7649999856948853, 'rewards/judge_quality/std': 0.2979933023452759, 'rewards/total_composite/mean': 0.7529339790344238, 'rewards/total_composite/std': 0.16097164154052734, 'reward': 0.7529339790344238, 'reward_std': 0.16097164154052734, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.15589402616024017, 'sampling/sampling_logp_difference/max': 1.3472023010253906, 'sampling/importance_sampling_ratio/min': 0.25996655225753784, 'sampling/importance_sampling_ratio/mean': 1.007192850112915, 'sampling/importance_sampling_ratio/max': 1.6380813121795654, 'entropy': 1.3930776715278625, 'clip_ratio/low_mean': 0.04749920126050711, 'clip_ratio/low_min': 0.04749920126050711, 'clip_ratio/high_mean': 0.09122474864125252, 'clip_ratio/high_max': 0.09122474864125252, 'clip_ratio/region_mean': 0.13872394990175962, 'reward_total_mean': 0.7529339790344238, 'reward_meter_mean': 0.6159642934799194, 'reward_meter_std': 0.4214301109313965, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9624999761581421, 'reward_repeat_soft_std': 0.0, 'reward_judge_quality_mean': 0.7649999856948853, 'reward_judge_quality_std': 0.2979933023452759, 'reward_total_composite_mean': 0.7529339790344238, 'reward_total_composite_std': 0.16097164154052734, 'epoch': 0.02} + 7%|▋ | 234/3300 [32:11<7:26:51, 8.74s/it]INFO 04-12 22:46:39 [block_pool.py:378] Successfully reset prefix cache + 7%|▋ | 235/3300 [32:18<7:02:15, 8.27s/it]2026-04-12 22:46:45,943 | INFO | train_grpo_train | metrics_logged mode=train step=235 + {'loss': 0.1022, 'grad_norm': 8.733047485351562, 'learning_rate': 9.29090909090909e-06, 'num_tokens': 456698.0, 'completions/mean_length': 109.375, 'completions/min_length': 85.0, 'completions/max_length': 139.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 109.375, 'completions/min_terminated_length': 85.0, 'completions/max_terminated_length': 139.0, 'rewards/meter/mean': 0.7656432390213013, 'rewards/meter/std': 0.26865702867507935, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9946377873420715, 'rewards/repeat_soft/std': 0.00715269148349762, 'rewards/judge_quality/mean': 0.3987500071525574, 'rewards/judge_quality/std': 0.06010407209396362, 'rewards/total_composite/mean': 0.7136281728744507, 'rewards/total_composite/std': 0.12620843946933746, 'reward': 0.7136281728744507, 'reward_std': 0.12620843946933746, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.21398591995239258, 'sampling/sampling_logp_difference/max': 1.3409252166748047, 'sampling/importance_sampling_ratio/min': 0.2616035044193268, 'sampling/importance_sampling_ratio/mean': 1.048514485359192, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.8638089895248413, 'clip_ratio/low_mean': 0.07135637197643518, 'clip_ratio/low_min': 0.07135637197643518, 'clip_ratio/high_mean': 0.0928946677595377, 'clip_ratio/high_max': 0.0928946677595377, 'clip_ratio/region_mean': 0.16425103973597288, 'reward_total_mean': 0.7136281728744507, 'reward_meter_mean': 0.7656432390213013, 'reward_meter_std': 0.26865702867507935, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9946377873420715, 'reward_repeat_soft_std': 0.00715269148349762, 'reward_judge_quality_mean': 0.3987500071525574, 'reward_judge_quality_std': 0.06010407209396362, 'reward_total_composite_mean': 0.7136281728744507, 'reward_total_composite_std': 0.12620843946933746, 'epoch': 0.02} + 7%|▋ | 235/3300 [32:18<7:02:15, 8.27s/it]INFO 04-12 22:46:46 [block_pool.py:378] Successfully reset prefix cache + 7%|▋ | 236/3300 [32:25<6:45:07, 7.93s/it]2026-04-12 22:46:53,154 | INFO | train_grpo_train | metrics_logged mode=train step=236 + {'loss': 0.034, 'grad_norm': 11.12470531463623, 'learning_rate': 9.28787878787879e-06, 'num_tokens': 459255.0, 'completions/mean_length': 125.625, 'completions/min_length': 96.0, 'completions/max_length': 144.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 125.625, 'completions/min_terminated_length': 96.0, 'completions/max_terminated_length': 144.0, 'rewards/meter/mean': 0.7550405859947205, 'rewards/meter/std': 0.19150368869304657, 'rewards/count_adherence/mean': 0.9791666269302368, 'rewards/count_adherence/std': 0.0589255727827549, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9707974195480347, 'rewards/repeat_soft/std': 0.057357292622327805, 'rewards/judge_quality/mean': 0.45749998092651367, 'rewards/judge_quality/std': 0.10606604069471359, 'rewards/total_composite/mean': 0.720973014831543, 'rewards/total_composite/std': 0.09331081807613373, 'reward': 0.720973014831543, 'reward_std': 0.09331081062555313, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.17728888988494873, 'sampling/sampling_logp_difference/max': 2.9329891204833984, 'sampling/importance_sampling_ratio/min': 0.0532376691699028, 'sampling/importance_sampling_ratio/mean': 1.0145037174224854, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.4931333884596825, 'clip_ratio/low_mean': 0.07065259478986263, 'clip_ratio/low_min': 0.07065259478986263, 'clip_ratio/high_mean': 0.11190121248364449, 'clip_ratio/high_max': 0.11190121248364449, 'clip_ratio/region_mean': 0.18255380727350712, 'reward_total_mean': 0.720973014831543, 'reward_meter_mean': 0.7550405859947205, 'reward_meter_std': 0.19150368869304657, 'reward_count_adherence_mean': 0.9791666269302368, 'reward_count_adherence_std': 0.0589255727827549, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9707974195480347, 'reward_repeat_soft_std': 0.057357292622327805, 'reward_judge_quality_mean': 0.45749998092651367, 'reward_judge_quality_std': 0.10606604069471359, 'reward_total_composite_mean': 0.720973014831543, 'reward_total_composite_std': 0.09331081807613373, 'epoch': 0.02} + 7%|▋ | 236/3300 [32:25<6:45:07, 7.93s/it]INFO 04-12 22:46:53 [block_pool.py:378] Successfully reset prefix cache + 7%|▋ | 237/3300 [32:32<6:25:42, 7.56s/it]2026-04-12 22:46:59,775 | INFO | train_grpo_train | metrics_logged mode=train step=237 + {'loss': 0.0483, 'grad_norm': 7.966315269470215, 'learning_rate': 9.284848484848485e-06, 'num_tokens': 461151.0, 'completions/mean_length': 83.0, 'completions/min_length': 78.0, 'completions/max_length': 86.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 83.0, 'completions/min_terminated_length': 78.0, 'completions/max_terminated_length': 86.0, 'rewards/meter/mean': 0.9735158085823059, 'rewards/meter/std': 0.02861112169921398, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9988561868667603, 'rewards/repeat_soft/std': 0.00124418328050524, 'rewards/judge_quality/mean': 0.4137499928474426, 'rewards/judge_quality/std': 0.20632413029670715, 'rewards/total_composite/mean': 0.8120927810668945, 'rewards/total_composite/std': 0.06512004882097244, 'reward': 0.8120927810668945, 'reward_std': 0.06512004137039185, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.17589770257472992, 'sampling/sampling_logp_difference/max': 2.1121904850006104, 'sampling/importance_sampling_ratio/min': 0.12097268551588058, 'sampling/importance_sampling_ratio/mean': 1.0163689851760864, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.5606450587511063, 'clip_ratio/low_mean': 0.07380836550146341, 'clip_ratio/low_min': 0.07380836550146341, 'clip_ratio/high_mean': 0.09933180175721645, 'clip_ratio/high_max': 0.09933180175721645, 'clip_ratio/region_mean': 0.17314016725867987, 'reward_total_mean': 0.8120927810668945, 'reward_meter_mean': 0.9735158085823059, 'reward_meter_std': 0.02861112169921398, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9988561868667603, 'reward_repeat_soft_std': 0.00124418328050524, 'reward_judge_quality_mean': 0.4137499928474426, 'reward_judge_quality_std': 0.20632413029670715, 'reward_total_composite_mean': 0.8120927810668945, 'reward_total_composite_std': 0.06512004882097244, 'epoch': 0.02} + 7%|▋ | 237/3300 [32:32<6:25:42, 7.56s/it]INFO 04-12 22:47:00 [block_pool.py:378] Successfully reset prefix cache + 7%|▋ | 238/3300 [32:41<6:48:43, 8.01s/it]2026-04-12 22:47:08,841 | INFO | train_grpo_train | metrics_logged mode=train step=238 + {'loss': 0.8318, 'grad_norm': 13.187271118164062, 'learning_rate': 9.281818181818183e-06, 'num_tokens': 463139.0, 'completions/mean_length': 91.5, 'completions/min_length': 54.0, 'completions/max_length': 321.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 91.5, 'completions/min_terminated_length': 54.0, 'completions/max_terminated_length': 321.0, 'rewards/meter/mean': 0.7172454595565796, 'rewards/meter/std': 0.4272644519805908, 'rewards/count_adherence/mean': 0.9375, 'rewards/count_adherence/std': 0.1767766922712326, 'rewards/hard_gate/mean': 0.875, 'rewards/hard_gate/std': 0.3535533845424652, 'rewards/repeat_soft/mean': 0.9606211185455322, 'rewards/repeat_soft/std': 0.09108290821313858, 'rewards/judge_quality/mean': 0.42250001430511475, 'rewards/judge_quality/std': 0.18108798563480377, 'rewards/total_composite/mean': 0.6625491976737976, 'rewards/total_composite/std': 0.30766403675079346, 'reward': 0.6625491976737976, 'reward_std': 0.30766406655311584, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.21388381719589233, 'sampling/sampling_logp_difference/max': 1.1890268325805664, 'sampling/importance_sampling_ratio/min': 0.3045174479484558, 'sampling/importance_sampling_ratio/mean': 1.0542694330215454, 'sampling/importance_sampling_ratio/max': 1.8876317739486694, 'entropy': 3.028022885322571, 'clip_ratio/low_mean': 0.03792402148246765, 'clip_ratio/low_min': 0.03792402148246765, 'clip_ratio/high_mean': 0.1028674142435193, 'clip_ratio/high_max': 0.1028674142435193, 'clip_ratio/region_mean': 0.14079143572598696, 'reward_total_mean': 0.6625491976737976, 'reward_meter_mean': 0.7172454595565796, 'reward_meter_std': 0.4272644519805908, 'reward_count_adherence_mean': 0.9375, 'reward_count_adherence_std': 0.1767766922712326, 'reward_hard_gate_mean': 0.875, 'reward_hard_gate_std': 0.3535533845424652, 'reward_repeat_soft_mean': 0.9606211185455322, 'reward_repeat_soft_std': 0.09108290821313858, 'reward_judge_quality_mean': 0.42250001430511475, 'reward_judge_quality_std': 0.18108798563480377, 'reward_total_composite_mean': 0.6625491976737976, 'reward_total_composite_std': 0.30766403675079346, 'epoch': 0.02} + 7%|▋ | 238/3300 [32:41<6:48:43, 8.01s/it]INFO 04-12 22:47:09 [block_pool.py:378] Successfully reset prefix cache + 7%|▋ | 239/3300 [32:47<6:26:10, 7.57s/it]2026-04-12 22:47:15,402 | INFO | train_grpo_train | metrics_logged mode=train step=239 + {'loss': 0.0302, 'grad_norm': 13.273502349853516, 'learning_rate': 9.27878787878788e-06, 'num_tokens': 464886.0, 'completions/mean_length': 55.375, 'completions/min_length': 42.0, 'completions/max_length': 63.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 55.375, 'completions/min_terminated_length': 42.0, 'completions/max_terminated_length': 63.0, 'rewards/meter/mean': 0.6486506462097168, 'rewards/meter/std': 0.39530149102211, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.996336817741394, 'rewards/repeat_soft/std': 0.006857405882328749, 'rewards/judge_quality/mean': 0.5024999976158142, 'rewards/judge_quality/std': 0.2681550681591034, 'rewards/total_composite/mean': 0.6922765374183655, 'rewards/total_composite/std': 0.21277062594890594, 'reward': 0.6922765374183655, 'reward_std': 0.21277062594890594, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.20548546314239502, 'sampling/sampling_logp_difference/max': 1.4224605560302734, 'sampling/importance_sampling_ratio/min': 0.24111999571323395, 'sampling/importance_sampling_ratio/mean': 1.0364271402359009, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.1999662667512894, 'clip_ratio/low_mean': 0.05759575683623552, 'clip_ratio/low_min': 0.05759575683623552, 'clip_ratio/high_mean': 0.13599523156881332, 'clip_ratio/high_max': 0.13599523156881332, 'clip_ratio/region_mean': 0.19359098840504885, 'reward_total_mean': 0.6922765374183655, 'reward_meter_mean': 0.6486506462097168, 'reward_meter_std': 0.39530149102211, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.996336817741394, 'reward_repeat_soft_std': 0.006857405882328749, 'reward_judge_quality_mean': 0.5024999976158142, 'reward_judge_quality_std': 0.2681550681591034, 'reward_total_composite_mean': 0.6922765374183655, 'reward_total_composite_std': 0.21277062594890594, 'epoch': 0.02} + 7%|▋ | 239/3300 [32:47<6:26:10, 7.57s/it]INFO 04-12 22:47:15 [block_pool.py:378] Successfully reset prefix cache + 7%|▋ | 240/3300 [32:55<6:21:52, 7.49s/it]2026-04-12 22:47:22,741 | INFO | train_grpo_train | metrics_logged mode=train step=240 + {'loss': -0.0149, 'grad_norm': 11.261079788208008, 'learning_rate': 9.275757575757577e-06, 'num_tokens': 466658.0, 'completions/mean_length': 64.5, 'completions/min_length': 56.0, 'completions/max_length': 71.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 64.5, 'completions/min_terminated_length': 56.0, 'completions/max_terminated_length': 71.0, 'rewards/meter/mean': 0.8303419351577759, 'rewards/meter/std': 0.2997992932796478, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9979175329208374, 'rewards/repeat_soft/std': 0.004616478458046913, 'rewards/judge_quality/mean': 0.35999998450279236, 'rewards/judge_quality/std': 0.09165150672197342, 'rewards/total_composite/mean': 0.7314456105232239, 'rewards/total_composite/std': 0.15546607971191406, 'reward': 0.7314456105232239, 'reward_std': 0.15546607971191406, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.212741881608963, 'sampling/sampling_logp_difference/max': 1.484156608581543, 'sampling/importance_sampling_ratio/min': 0.22669345140457153, 'sampling/importance_sampling_ratio/mean': 1.0417413711547852, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.943022519350052, 'clip_ratio/low_mean': 0.038370998576283455, 'clip_ratio/low_min': 0.038370998576283455, 'clip_ratio/high_mean': 0.15426747128367424, 'clip_ratio/high_max': 0.15426747128367424, 'clip_ratio/region_mean': 0.1926384698599577, 'reward_total_mean': 0.7314456105232239, 'reward_meter_mean': 0.8303419351577759, 'reward_meter_std': 0.2997992932796478, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9979175329208374, 'reward_repeat_soft_std': 0.004616478458046913, 'reward_judge_quality_mean': 0.35999998450279236, 'reward_judge_quality_std': 0.09165150672197342, 'reward_total_composite_mean': 0.7314456105232239, 'reward_total_composite_std': 0.15546607971191406, 'epoch': 0.02} + 7%|▋ | 240/3300 [32:55<6:21:52, 7.49s/it]INFO 04-12 22:47:22 [block_pool.py:378] Successfully reset prefix cache + 7%|▋ | 241/3300 [33:03<6:34:52, 7.75s/it]2026-04-12 22:47:31,043 | INFO | train_grpo_train | metrics_logged mode=train step=241 + {'loss': -0.0398, 'grad_norm': 10.336009979248047, 'learning_rate': 9.272727272727273e-06, 'num_tokens': 468638.0, 'completions/mean_length': 84.5, 'completions/min_length': 67.0, 'completions/max_length': 97.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 84.5, 'completions/min_terminated_length': 67.0, 'completions/max_terminated_length': 97.0, 'rewards/meter/mean': 0.6151079535484314, 'rewards/meter/std': 0.3954385221004486, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9956986904144287, 'rewards/repeat_soft/std': 0.004514547996222973, 'rewards/judge_quality/mean': 0.5237500071525574, 'rewards/judge_quality/std': 0.25150617957115173, 'rewards/total_composite/mean': 0.6834934949874878, 'rewards/total_composite/std': 0.16889327764511108, 'reward': 0.6834934949874878, 'reward_std': 0.16889327764511108, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.20427070558071136, 'sampling/sampling_logp_difference/max': 1.922989845275879, 'sampling/importance_sampling_ratio/min': 0.1461692899465561, 'sampling/importance_sampling_ratio/mean': 1.037583589553833, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.4143103063106537, 'clip_ratio/low_mean': 0.04924237309023738, 'clip_ratio/low_min': 0.04924237309023738, 'clip_ratio/high_mean': 0.11960223410278559, 'clip_ratio/high_max': 0.11960223410278559, 'clip_ratio/region_mean': 0.16884460719302297, 'reward_total_mean': 0.6834934949874878, 'reward_meter_mean': 0.6151079535484314, 'reward_meter_std': 0.3954385221004486, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9956986904144287, 'reward_repeat_soft_std': 0.004514547996222973, 'reward_judge_quality_mean': 0.5237500071525574, 'reward_judge_quality_std': 0.25150617957115173, 'reward_total_composite_mean': 0.6834934949874878, 'reward_total_composite_std': 0.16889327764511108, 'epoch': 0.02} + 7%|▋ | 241/3300 [33:03<6:34:52, 7.75s/it]INFO 04-12 22:47:31 [block_pool.py:378] Successfully reset prefix cache + 7%|▋ | 242/3300 [33:09<6:15:18, 7.36s/it]2026-04-12 22:47:37,504 | INFO | train_grpo_train | metrics_logged mode=train step=242 + {'loss': 0.0356, 'grad_norm': 28.52313995361328, 'learning_rate': 9.26969696969697e-06, 'num_tokens': 470079.0, 'completions/mean_length': 23.125, 'completions/min_length': 19.0, 'completions/max_length': 27.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 23.125, 'completions/min_terminated_length': 19.0, 'completions/max_terminated_length': 27.0, 'rewards/meter/mean': 0.498305082321167, 'rewards/meter/std': 0.46030333638191223, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9624999761581421, 'rewards/repeat_soft/std': 0.0, 'rewards/judge_quality/mean': 0.3512499928474426, 'rewards/judge_quality/std': 0.11630470305681229, 'rewards/total_composite/mean': 0.5758622884750366, 'rewards/total_composite/std': 0.23587921261787415, 'reward': 0.5758622884750366, 'reward_std': 0.23587918281555176, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.14344589412212372, 'sampling/sampling_logp_difference/max': 2.1663403511047363, 'sampling/importance_sampling_ratio/min': 0.11459623277187347, 'sampling/importance_sampling_ratio/mean': 0.9992484450340271, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 0.8322795741260052, 'clip_ratio/low_mean': 0.05294940201565623, 'clip_ratio/low_min': 0.05294940201565623, 'clip_ratio/high_mean': 0.038419914431869984, 'clip_ratio/high_max': 0.038419914431869984, 'clip_ratio/region_mean': 0.09136931644752622, 'reward_total_mean': 0.5758622884750366, 'reward_meter_mean': 0.498305082321167, 'reward_meter_std': 0.46030333638191223, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9624999761581421, 'reward_repeat_soft_std': 0.0, 'reward_judge_quality_mean': 0.3512499928474426, 'reward_judge_quality_std': 0.11630470305681229, 'reward_total_composite_mean': 0.5758622884750366, 'reward_total_composite_std': 0.23587921261787415, 'epoch': 0.02} + 7%|▋ | 242/3300 [33:09<6:15:18, 7.36s/it]INFO 04-12 22:47:37 [block_pool.py:378] Successfully reset prefix cache + 7%|▋ | 243/3300 [33:16<6:07:40, 7.22s/it]2026-04-12 22:47:44,377 | INFO | train_grpo_train | metrics_logged mode=train step=243 + {'loss': 0.0722, 'grad_norm': 10.151678085327148, 'learning_rate': 9.266666666666667e-06, 'num_tokens': 472070.0, 'completions/mean_length': 85.875, 'completions/min_length': 78.0, 'completions/max_length': 98.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 85.875, 'completions/min_terminated_length': 78.0, 'completions/max_terminated_length': 98.0, 'rewards/meter/mean': 0.5651149749755859, 'rewards/meter/std': 0.4197004735469818, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.991955578327179, 'rewards/repeat_soft/std': 0.008437552489340305, 'rewards/judge_quality/mean': 0.3774999976158142, 'rewards/judge_quality/std': 0.07869470119476318, 'rewards/total_composite/mean': 0.6167473196983337, 'rewards/total_composite/std': 0.19320012629032135, 'reward': 0.6167473196983337, 'reward_std': 0.19320012629032135, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.18541502952575684, 'sampling/sampling_logp_difference/max': 1.4706039428710938, 'sampling/importance_sampling_ratio/min': 0.22978666424751282, 'sampling/importance_sampling_ratio/mean': 1.038940191268921, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.329676851630211, 'clip_ratio/low_mean': 0.08723794110119343, 'clip_ratio/low_min': 0.08723794110119343, 'clip_ratio/high_mean': 0.08647382818162441, 'clip_ratio/high_max': 0.08647382818162441, 'clip_ratio/region_mean': 0.17371176928281784, 'reward_total_mean': 0.6167473196983337, 'reward_meter_mean': 0.5651149749755859, 'reward_meter_std': 0.4197004735469818, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.991955578327179, 'reward_repeat_soft_std': 0.008437552489340305, 'reward_judge_quality_mean': 0.3774999976158142, 'reward_judge_quality_std': 0.07869470119476318, 'reward_total_composite_mean': 0.6167473196983337, 'reward_total_composite_std': 0.19320012629032135, 'epoch': 0.02} + 7%|▋ | 243/3300 [33:16<6:07:40, 7.22s/it]INFO 04-12 22:47:44 [block_pool.py:378] Successfully reset prefix cache + 7%|▋ | 244/3300 [33:24<6:11:32, 7.29s/it]2026-04-12 22:47:51,852 | INFO | train_grpo_train | metrics_logged mode=train step=244 + {'loss': 0.0696, 'grad_norm': 11.594178199768066, 'learning_rate': 9.263636363636364e-06, 'num_tokens': 473719.0, 'completions/mean_length': 58.125, 'completions/min_length': 46.0, 'completions/max_length': 68.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 58.125, 'completions/min_terminated_length': 46.0, 'completions/max_terminated_length': 68.0, 'rewards/meter/mean': 0.24202539026737213, 'rewards/meter/std': 0.2553742825984955, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9992761611938477, 'rewards/repeat_soft/std': 0.001410387922078371, 'rewards/judge_quality/mean': 0.38499999046325684, 'rewards/judge_quality/std': 0.0843462198972702, 'rewards/total_composite/mean': 0.4743390679359436, 'rewards/total_composite/std': 0.11051017045974731, 'reward': 0.4743390679359436, 'reward_std': 0.11051014810800552, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.20545533299446106, 'sampling/sampling_logp_difference/max': 1.0835037231445312, 'sampling/importance_sampling_ratio/min': 0.3384077548980713, 'sampling/importance_sampling_ratio/mean': 1.0498751401901245, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.7120381891727448, 'clip_ratio/low_mean': 0.09280562400817871, 'clip_ratio/low_min': 0.09280562400817871, 'clip_ratio/high_mean': 0.06330752745270729, 'clip_ratio/high_max': 0.06330752745270729, 'clip_ratio/region_mean': 0.156113151460886, 'reward_total_mean': 0.4743390679359436, 'reward_meter_mean': 0.24202539026737213, 'reward_meter_std': 0.2553742825984955, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9992761611938477, 'reward_repeat_soft_std': 0.001410387922078371, 'reward_judge_quality_mean': 0.38499999046325684, 'reward_judge_quality_std': 0.0843462198972702, 'reward_total_composite_mean': 0.4743390679359436, 'reward_total_composite_std': 0.11051017045974731, 'epoch': 0.02} + 7%|▋ | 244/3300 [33:24<6:11:32, 7.29s/it]INFO 04-12 22:47:52 [block_pool.py:378] Successfully reset prefix cache + 7%|▋ | 245/3300 [33:30<5:58:57, 7.05s/it]2026-04-12 22:47:58,345 | INFO | train_grpo_train | metrics_logged mode=train step=245 + {'loss': 0.0455, 'grad_norm': 12.824295043945312, 'learning_rate': 9.260606060606062e-06, 'num_tokens': 475562.0, 'completions/mean_length': 56.375, 'completions/min_length': 43.0, 'completions/max_length': 70.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 56.375, 'completions/min_terminated_length': 43.0, 'completions/max_terminated_length': 70.0, 'rewards/meter/mean': 0.783051609992981, 'rewards/meter/std': 0.28790801763534546, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 0.875, 'rewards/hard_gate/std': 0.3535533845424652, 'rewards/repeat_soft/mean': 0.9996891021728516, 'rewards/repeat_soft/std': 0.0006069060764275491, 'rewards/judge_quality/mean': 0.36374998092651367, 'rewards/judge_quality/std': 0.09500939399003983, 'rewards/total_composite/mean': 0.6594811677932739, 'rewards/total_composite/std': 0.2779873311519623, 'reward': 0.6594811677932739, 'reward_std': 0.2779873311519623, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.20442284643650055, 'sampling/sampling_logp_difference/max': 1.208592414855957, 'sampling/importance_sampling_ratio/min': 0.29861733317375183, 'sampling/importance_sampling_ratio/mean': 1.0455328226089478, 'sampling/importance_sampling_ratio/max': 1.9741625785827637, 'entropy': 2.409978613257408, 'clip_ratio/low_mean': 0.048394243232905865, 'clip_ratio/low_min': 0.048394243232905865, 'clip_ratio/high_mean': 0.136877179145813, 'clip_ratio/high_max': 0.136877179145813, 'clip_ratio/region_mean': 0.18527142237871885, 'reward_total_mean': 0.6594811677932739, 'reward_meter_mean': 0.783051609992981, 'reward_meter_std': 0.28790801763534546, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 0.875, 'reward_hard_gate_std': 0.3535533845424652, 'reward_repeat_soft_mean': 0.9996891021728516, 'reward_repeat_soft_std': 0.0006069060764275491, 'reward_judge_quality_mean': 0.36374998092651367, 'reward_judge_quality_std': 0.09500939399003983, 'reward_total_composite_mean': 0.6594811677932739, 'reward_total_composite_std': 0.2779873311519623, 'epoch': 0.02} + 7%|▋ | 245/3300 [33:30<5:58:57, 7.05s/it]INFO 04-12 22:47:58 [block_pool.py:378] Successfully reset prefix cache + 7%|▋ | 246/3300 [33:41<6:48:48, 8.03s/it]2026-04-12 22:48:08,654 | INFO | train_grpo_train | metrics_logged mode=train step=246 + {'loss': -0.0128, 'grad_norm': 5.772524833679199, 'learning_rate': 9.257575757575759e-06, 'num_tokens': 478739.0, 'completions/mean_length': 201.125, 'completions/min_length': 147.0, 'completions/max_length': 269.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 201.125, 'completions/min_terminated_length': 147.0, 'completions/max_terminated_length': 269.0, 'rewards/meter/mean': 0.43260225653648376, 'rewards/meter/std': 0.3558891713619232, 'rewards/count_adherence/mean': 0.8125, 'rewards/count_adherence/std': 0.10681164264678955, 'rewards/hard_gate/mean': 0.875, 'rewards/hard_gate/std': 0.3535533845424652, 'rewards/repeat_soft/mean': 0.9973302483558655, 'rewards/repeat_soft/std': 0.003060556249693036, 'rewards/judge_quality/mean': 0.25874999165534973, 'rewards/judge_quality/std': 0.07395702600479126, 'rewards/total_composite/mean': 0.4241829514503479, 'rewards/total_composite/std': 0.23961442708969116, 'reward': 0.4241829514503479, 'reward_std': 0.23961441218852997, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.21506229043006897, 'sampling/sampling_logp_difference/max': 1.452444076538086, 'sampling/importance_sampling_ratio/min': 0.23399768769741058, 'sampling/importance_sampling_ratio/mean': 1.0483766794204712, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 3.610942929983139, 'clip_ratio/low_mean': 0.05695440340787172, 'clip_ratio/low_min': 0.05695440340787172, 'clip_ratio/high_mean': 0.12621658854186535, 'clip_ratio/high_max': 0.12621658854186535, 'clip_ratio/region_mean': 0.18317099194973707, 'reward_total_mean': 0.4241829514503479, 'reward_meter_mean': 0.43260225653648376, 'reward_meter_std': 0.3558891713619232, 'reward_count_adherence_mean': 0.8125, 'reward_count_adherence_std': 0.10681164264678955, 'reward_hard_gate_mean': 0.875, 'reward_hard_gate_std': 0.3535533845424652, 'reward_repeat_soft_mean': 0.9973302483558655, 'reward_repeat_soft_std': 0.003060556249693036, 'reward_judge_quality_mean': 0.25874999165534973, 'reward_judge_quality_std': 0.07395702600479126, 'reward_total_composite_mean': 0.4241829514503479, 'reward_total_composite_std': 0.23961442708969116, 'epoch': 0.02} + 7%|▋ | 246/3300 [33:41<6:48:48, 8.03s/it]INFO 04-12 22:48:08 [block_pool.py:378] Successfully reset prefix cache + 7%|▋ | 247/3300 [33:48<6:38:44, 7.84s/it]2026-04-12 22:48:16,036 | INFO | train_grpo_train | metrics_logged mode=train step=247 + {'loss': -0.0354, 'grad_norm': 9.905174255371094, 'learning_rate': 9.254545454545454e-06, 'num_tokens': 480554.0, 'completions/mean_length': 66.875, 'completions/min_length': 50.0, 'completions/max_length': 106.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 66.875, 'completions/min_terminated_length': 50.0, 'completions/max_terminated_length': 106.0, 'rewards/meter/mean': 0.3794197738170624, 'rewards/meter/std': 0.2575729191303253, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9990973472595215, 'rewards/repeat_soft/std': 0.0019004953792318702, 'rewards/judge_quality/mean': 0.3987500071525574, 'rewards/judge_quality/std': 0.06010407209396362, 'rewards/total_composite/mean': 0.5402736067771912, 'rewards/total_composite/std': 0.12342807650566101, 'reward': 0.5402736067771912, 'reward_std': 0.12342805415391922, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.20863491296768188, 'sampling/sampling_logp_difference/max': 1.5102834701538086, 'sampling/importance_sampling_ratio/min': 0.22084736824035645, 'sampling/importance_sampling_ratio/mean': 1.0235635042190552, 'sampling/importance_sampling_ratio/max': 1.8423658609390259, 'entropy': 2.909391537308693, 'clip_ratio/low_mean': 0.07251503877341747, 'clip_ratio/low_min': 0.07251503877341747, 'clip_ratio/high_mean': 0.12647932302206755, 'clip_ratio/high_max': 0.12647932302206755, 'clip_ratio/region_mean': 0.19899436179548502, 'reward_total_mean': 0.5402736067771912, 'reward_meter_mean': 0.3794197738170624, 'reward_meter_std': 0.2575729191303253, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9990973472595215, 'reward_repeat_soft_std': 0.0019004953792318702, 'reward_judge_quality_mean': 0.3987500071525574, 'reward_judge_quality_std': 0.06010407209396362, 'reward_total_composite_mean': 0.5402736067771912, 'reward_total_composite_std': 0.12342807650566101, 'epoch': 0.02} + 7%|▋ | 247/3300 [33:48<6:38:44, 7.84s/it]INFO 04-12 22:48:16 [block_pool.py:378] Successfully reset prefix cache + 8%|▊ | 248/3300 [33:55<6:27:20, 7.61s/it]2026-04-12 22:48:23,145 | INFO | train_grpo_train | metrics_logged mode=train step=248 + {'loss': -0.0216, 'grad_norm': 8.936887741088867, 'learning_rate': 9.251515151515152e-06, 'num_tokens': 482600.0, 'completions/mean_length': 95.75, 'completions/min_length': 64.0, 'completions/max_length': 121.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 95.75, 'completions/min_terminated_length': 64.0, 'completions/max_terminated_length': 121.0, 'rewards/meter/mean': 0.413760244846344, 'rewards/meter/std': 0.2599477469921112, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9972332715988159, 'rewards/repeat_soft/std': 0.003570000873878598, 'rewards/judge_quality/mean': 0.3987500071525574, 'rewards/judge_quality/std': 0.06010407209396362, 'rewards/total_composite/mean': 0.5555404424667358, 'rewards/total_composite/std': 0.12218713760375977, 'reward': 0.5555404424667358, 'reward_std': 0.12218714505434036, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.22171355783939362, 'sampling/sampling_logp_difference/max': 1.506246566772461, 'sampling/importance_sampling_ratio/min': 0.2217407077550888, 'sampling/importance_sampling_ratio/mean': 1.0438802242279053, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.608716830611229, 'clip_ratio/low_mean': 0.12864871509373188, 'clip_ratio/low_min': 0.12864871509373188, 'clip_ratio/high_mean': 0.06282605789601803, 'clip_ratio/high_max': 0.06282605789601803, 'clip_ratio/region_mean': 0.1914747729897499, 'reward_total_mean': 0.5555404424667358, 'reward_meter_mean': 0.413760244846344, 'reward_meter_std': 0.2599477469921112, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9972332715988159, 'reward_repeat_soft_std': 0.003570000873878598, 'reward_judge_quality_mean': 0.3987500071525574, 'reward_judge_quality_std': 0.06010407209396362, 'reward_total_composite_mean': 0.5555404424667358, 'reward_total_composite_std': 0.12218713760375977, 'epoch': 0.02} + 8%|▊ | 248/3300 [33:55<6:27:20, 7.61s/it]INFO 04-12 22:48:23 [block_pool.py:378] Successfully reset prefix cache + 8%|▊ | 249/3300 [34:02<6:09:52, 7.27s/it]2026-04-12 22:48:29,639 | INFO | train_grpo_train | metrics_logged mode=train step=249 + {'loss': 0.0716, 'grad_norm': 18.126298904418945, 'learning_rate': 9.248484848484849e-06, 'num_tokens': 484116.0, 'completions/mean_length': 44.5, 'completions/min_length': 40.0, 'completions/max_length': 48.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 44.5, 'completions/min_terminated_length': 40.0, 'completions/max_terminated_length': 48.0, 'rewards/meter/mean': 0.49406689405441284, 'rewards/meter/std': 0.3191210925579071, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9943512678146362, 'rewards/repeat_soft/std': 0.008807245641946793, 'rewards/judge_quality/mean': 0.6075000166893005, 'rewards/judge_quality/std': 0.25877460837364197, 'rewards/total_composite/mean': 0.6540151834487915, 'rewards/total_composite/std': 0.17928293347358704, 'reward': 0.6540151834487915, 'reward_std': 0.17928291857242584, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.1901768445968628, 'sampling/sampling_logp_difference/max': 1.7334613800048828, 'sampling/importance_sampling_ratio/min': 0.1766718178987503, 'sampling/importance_sampling_ratio/mean': 1.0153061151504517, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.0152252092957497, 'clip_ratio/low_mean': 0.07058080844581127, 'clip_ratio/low_min': 0.07058080844581127, 'clip_ratio/high_mean': 0.09148098807781935, 'clip_ratio/high_max': 0.09148098807781935, 'clip_ratio/region_mean': 0.16206179652363062, 'reward_total_mean': 0.6540151834487915, 'reward_meter_mean': 0.49406689405441284, 'reward_meter_std': 0.3191210925579071, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9943512678146362, 'reward_repeat_soft_std': 0.008807245641946793, 'reward_judge_quality_mean': 0.6075000166893005, 'reward_judge_quality_std': 0.25877460837364197, 'reward_total_composite_mean': 0.6540151834487915, 'reward_total_composite_std': 0.17928293347358704, 'epoch': 0.03} + 8%|▊ | 249/3300 [34:02<6:09:52, 7.27s/it]INFO 04-12 22:48:29 [block_pool.py:378] Successfully reset prefix cache + 8%|▊ | 250/3300 [34:10<6:23:14, 7.54s/it]2026-04-12 22:48:37,799 | INFO | train_grpo_train | metrics_logged mode=train step=250 + {'loss': 0.0403, 'grad_norm': 6.014837265014648, 'learning_rate': 9.245454545454546e-06, 'num_tokens': 487259.0, 'completions/mean_length': 186.875, 'completions/min_length': 132.0, 'completions/max_length': 213.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 186.875, 'completions/min_terminated_length': 132.0, 'completions/max_terminated_length': 213.0, 'rewards/meter/mean': 0.6916961073875427, 'rewards/meter/std': 0.2751163840293884, 'rewards/count_adherence/mean': 0.800000011920929, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.994820773601532, 'rewards/repeat_soft/std': 0.005338526330888271, 'rewards/judge_quality/mean': 0.35624998807907104, 'rewards/judge_quality/std': 0.08798335492610931, 'rewards/total_composite/mean': 0.6376203298568726, 'rewards/total_composite/std': 0.12483447045087814, 'reward': 0.6376203298568726, 'reward_std': 0.12483445554971695, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.2097940295934677, 'sampling/sampling_logp_difference/max': 1.2559013366699219, 'sampling/importance_sampling_ratio/min': 0.28481900691986084, 'sampling/importance_sampling_ratio/mean': 1.0570787191390991, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 3.3793243765830994, 'clip_ratio/low_mean': 0.07245279382914305, 'clip_ratio/low_min': 0.07245279382914305, 'clip_ratio/high_mean': 0.08382073976099491, 'clip_ratio/high_max': 0.08382073976099491, 'clip_ratio/region_mean': 0.15627353359013796, 'reward_total_mean': 0.6376203298568726, 'reward_meter_mean': 0.6916961073875427, 'reward_meter_std': 0.2751163840293884, 'reward_count_adherence_mean': 0.800000011920929, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.994820773601532, 'reward_repeat_soft_std': 0.005338526330888271, 'reward_judge_quality_mean': 0.35624998807907104, 'reward_judge_quality_std': 0.08798335492610931, 'reward_total_composite_mean': 0.6376203298568726, 'reward_total_composite_std': 0.12483447045087814, 'epoch': 0.03} + 8%|▊ | 250/3300 [34:10<6:23:14, 7.54s/it]INFO 04-12 22:48:38 [block_pool.py:378] Successfully reset prefix cache +/root/workspace/Shaer/grpo/.venv/lib/python3.11/site-packages/trl/trainer/grpo_trainer.py:1450: UserWarning: std(): degrees of freedom is <= 0. Correction should be strictly less than the reduction factor (input numel divided by output numel). (Triggered internally at /pytorch/aten/src/ATen/native/ReduceOps.cpp:1839.) + std_rewards = rewards.view(-1, self.num_generations).std(dim=1) + + 0%| | 0/10 [00:00