diff --git "a/train_stdout.log" "b/train_stdout.log" --- "a/train_stdout.log" +++ "b/train_stdout.log" @@ -399,3 +399,169 @@ The tokenizer has new PAD/BOS/EOS tokens that differ from the model config and g 3%|▎ | 101/3300 [14:19<21:31:50, 24.23s/it]2026-04-12 22:28:46,892 | INFO | train_grpo_train | metrics_logged mode=train step=101 {'loss': 0.0307, 'grad_norm': 8.720218658447266, 'learning_rate': 9.696969696969698e-06, 'num_tokens': 192271.0, 'completions/mean_length': 96.375, 'completions/min_length': 83.0, 'completions/max_length': 104.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 96.375, 'completions/min_terminated_length': 83.0, 'completions/max_terminated_length': 104.0, 'rewards/meter/mean': 0.9825379252433777, 'rewards/meter/std': 0.03238074481487274, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9823436737060547, 'rewards/repeat_soft/std': 0.021364767104387283, 'rewards/judge_quality/mean': 0.3987500071525574, 'rewards/judge_quality/std': 0.06010407209396362, 'rewards/total_composite/mean': 0.8100014328956604, 'rewards/total_composite/std': 0.02203506976366043, 'reward': 0.8100014328956604, 'reward_std': 0.02203506790101528, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.19199949502944946, 'sampling/sampling_logp_difference/max': 1.4776219129562378, 'sampling/importance_sampling_ratio/min': 0.2281796783208847, 'sampling/importance_sampling_ratio/mean': 1.0370380878448486, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.443674087524414, 'clip_ratio/low_mean': 0.04417553171515465, 'clip_ratio/low_min': 0.04417553171515465, 'clip_ratio/high_mean': 0.1387271974235773, 'clip_ratio/high_max': 0.1387271974235773, 'clip_ratio/region_mean': 0.18290272913873196, 'reward_total_mean': 0.8100014328956604, 'reward_meter_mean': 0.9825379252433777, 'reward_meter_std': 0.03238074481487274, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9823436737060547, 'reward_repeat_soft_std': 0.021364767104387283, 'reward_judge_quality_mean': 0.3987500071525574, 'reward_judge_quality_std': 0.06010407209396362, 'reward_total_composite_mean': 0.8100014328956604, 'reward_total_composite_std': 0.02203506976366043, 'epoch': 0.01} 3%|▎ | 101/3300 [14:19<21:31:50, 24.23s/it]INFO 04-12 22:28:47 [block_pool.py:378] Successfully reset prefix cache + 3%|▎ | 102/3300 [14:30<18:03:09, 20.32s/it]2026-04-12 22:28:58,082 | INFO | train_grpo_train | metrics_logged mode=train step=102 + {'loss': 0.0103, 'grad_norm': 10.914584159851074, 'learning_rate': 9.693939393939395e-06, 'num_tokens': 194566.0, 'completions/mean_length': 107.875, 'completions/min_length': 84.0, 'completions/max_length': 141.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 107.875, 'completions/min_terminated_length': 84.0, 'completions/max_terminated_length': 141.0, 'rewards/meter/mean': 0.4683343768119812, 'rewards/meter/std': 0.3329935371875763, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9961557388305664, 'rewards/repeat_soft/std': 0.0035384881775826216, 'rewards/judge_quality/mean': 0.45749998092651367, 'rewards/judge_quality/std': 0.10606604069471359, 'rewards/total_composite/mean': 0.5976160168647766, 'rewards/total_composite/std': 0.17027634382247925, 'reward': 0.5976160168647766, 'reward_std': 0.17027634382247925, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.22297446429729462, 'sampling/sampling_logp_difference/max': 2.2250664234161377, 'sampling/importance_sampling_ratio/min': 0.10806024074554443, 'sampling/importance_sampling_ratio/mean': 1.0301787853240967, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.9477074295282364, 'clip_ratio/low_mean': 0.14782720245420933, 'clip_ratio/low_min': 0.14782720245420933, 'clip_ratio/high_mean': 0.08563361689448357, 'clip_ratio/high_max': 0.08563361689448357, 'clip_ratio/region_mean': 0.2334608193486929, 'reward_total_mean': 0.5976160168647766, 'reward_meter_mean': 0.4683343768119812, 'reward_meter_std': 0.3329935371875763, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9961557388305664, 'reward_repeat_soft_std': 0.0035384881775826216, 'reward_judge_quality_mean': 0.45749998092651367, 'reward_judge_quality_std': 0.10606604069471359, 'reward_total_composite_mean': 0.5976160168647766, 'reward_total_composite_std': 0.17027634382247925, 'epoch': 0.01} + 3%|▎ | 102/3300 [14:30<18:03:09, 20.32s/it]INFO 04-12 22:28:58 [block_pool.py:378] Successfully reset prefix cache + 3%|▎ | 103/3300 [14:40<15:23:40, 17.34s/it]2026-04-12 22:29:08,448 | INFO | train_grpo_train | metrics_logged mode=train step=103 + {'loss': 0.0549, 'grad_norm': 17.430255889892578, 'learning_rate': 9.690909090909092e-06, 'num_tokens': 196322.0, 'completions/mean_length': 62.5, 'completions/min_length': 59.0, 'completions/max_length': 67.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 62.5, 'completions/min_terminated_length': 59.0, 'completions/max_terminated_length': 67.0, 'rewards/meter/mean': 0.8165329694747925, 'rewards/meter/std': 0.3091726303100586, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9988213777542114, 'rewards/repeat_soft/std': 0.0023674487601965666, 'rewards/judge_quality/mean': 0.6187499761581421, 'rewards/judge_quality/std': 0.24976776540279388, 'rewards/total_composite/mean': 0.802946925163269, 'rewards/total_composite/std': 0.1676766276359558, 'reward': 0.802946925163269, 'reward_std': 0.1676766276359558, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.1588955819606781, 'sampling/sampling_logp_difference/max': 1.617774486541748, 'sampling/importance_sampling_ratio/min': 0.19833961129188538, 'sampling/importance_sampling_ratio/mean': 1.0459027290344238, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.1695427224040031, 'clip_ratio/low_mean': 0.02010148297995329, 'clip_ratio/low_min': 0.02010148297995329, 'clip_ratio/high_mean': 0.09783541085198522, 'clip_ratio/high_max': 0.09783541085198522, 'clip_ratio/region_mean': 0.1179368938319385, 'reward_total_mean': 0.802946925163269, 'reward_meter_mean': 0.8165329694747925, 'reward_meter_std': 0.3091726303100586, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9988213777542114, 'reward_repeat_soft_std': 0.0023674487601965666, 'reward_judge_quality_mean': 0.6187499761581421, 'reward_judge_quality_std': 0.24976776540279388, 'reward_total_composite_mean': 0.802946925163269, 'reward_total_composite_std': 0.1676766276359558, 'epoch': 0.01} + 3%|▎ | 103/3300 [14:40<15:23:40, 17.34s/it]INFO 04-12 22:29:08 [block_pool.py:378] Successfully reset prefix cache + 3%|▎ | 104/3300 [14:47<12:28:35, 14.05s/it]2026-04-12 22:29:14,844 | INFO | train_grpo_train | metrics_logged mode=train step=104 + {'loss': 0.0394, 'grad_norm': 15.631914138793945, 'learning_rate': 9.687878787878788e-06, 'num_tokens': 198005.0, 'completions/mean_length': 56.375, 'completions/min_length': 50.0, 'completions/max_length': 65.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 56.375, 'completions/min_terminated_length': 50.0, 'completions/max_terminated_length': 65.0, 'rewards/meter/mean': 0.9202769994735718, 'rewards/meter/std': 0.12859795987606049, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9968171715736389, 'rewards/repeat_soft/std': 0.00538486847653985, 'rewards/judge_quality/mean': 0.5600000023841858, 'rewards/judge_quality/std': 0.22258226573467255, 'rewards/total_composite/mean': 0.8318063616752625, 'rewards/total_composite/std': 0.05040104314684868, 'reward': 0.8318063616752625, 'reward_std': 0.05040103569626808, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.18321827054023743, 'sampling/sampling_logp_difference/max': 1.7233179807662964, 'sampling/importance_sampling_ratio/min': 0.17847299575805664, 'sampling/importance_sampling_ratio/mean': 1.009376049041748, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.4439412578940392, 'clip_ratio/low_mean': 0.13680845219641924, 'clip_ratio/low_min': 0.13680845219641924, 'clip_ratio/high_mean': 0.04409965127706528, 'clip_ratio/high_max': 0.04409965127706528, 'clip_ratio/region_mean': 0.18090810347348452, 'reward_total_mean': 0.8318063616752625, 'reward_meter_mean': 0.9202769994735718, 'reward_meter_std': 0.12859795987606049, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9968171715736389, 'reward_repeat_soft_std': 0.00538486847653985, 'reward_judge_quality_mean': 0.5600000023841858, 'reward_judge_quality_std': 0.22258226573467255, 'reward_total_composite_mean': 0.8318063616752625, 'reward_total_composite_std': 0.05040104314684868, 'epoch': 0.01} + 3%|▎ | 104/3300 [14:47<12:28:35, 14.05s/it]INFO 04-12 22:29:15 [block_pool.py:378] Successfully reset prefix cache + 3%|▎ | 105/3300 [14:54<10:36:21, 11.95s/it]2026-04-12 22:29:21,888 | INFO | train_grpo_train | metrics_logged mode=train step=105 + {'loss': -0.0037, 'grad_norm': 8.602232933044434, 'learning_rate': 9.684848484848487e-06, 'num_tokens': 200456.0, 'completions/mean_length': 122.375, 'completions/min_length': 105.0, 'completions/max_length': 138.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 122.375, 'completions/min_terminated_length': 105.0, 'completions/max_terminated_length': 138.0, 'rewards/meter/mean': 0.618858814239502, 'rewards/meter/std': 0.37144485116004944, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 0.875, 'rewards/hard_gate/std': 0.3535533845424652, 'rewards/repeat_soft/mean': 0.9973665475845337, 'rewards/repeat_soft/std': 0.0020187776535749435, 'rewards/judge_quality/mean': 0.39375001192092896, 'rewards/judge_quality/std': 0.15638209879398346, 'rewards/total_composite/mean': 0.5785918235778809, 'rewards/total_composite/std': 0.2889094352722168, 'reward': 0.5785918235778809, 'reward_std': 0.2889094650745392, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.18066643178462982, 'sampling/sampling_logp_difference/max': 1.9373149871826172, 'sampling/importance_sampling_ratio/min': 0.14409030973911285, 'sampling/importance_sampling_ratio/mean': 1.02179753780365, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.8448036909103394, 'clip_ratio/low_mean': 0.08731714449822903, 'clip_ratio/low_min': 0.08731714449822903, 'clip_ratio/high_mean': 0.13265781849622726, 'clip_ratio/high_max': 0.13265781849622726, 'clip_ratio/region_mean': 0.2199749629944563, 'reward_total_mean': 0.5785918235778809, 'reward_meter_mean': 0.618858814239502, 'reward_meter_std': 0.37144485116004944, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 0.875, 'reward_hard_gate_std': 0.3535533845424652, 'reward_repeat_soft_mean': 0.9973665475845337, 'reward_repeat_soft_std': 0.0020187776535749435, 'reward_judge_quality_mean': 0.39375001192092896, 'reward_judge_quality_std': 0.15638209879398346, 'reward_total_composite_mean': 0.5785918235778809, 'reward_total_composite_std': 0.2889094352722168, 'epoch': 0.01} + 3%|▎ | 105/3300 [14:54<10:36:21, 11.95s/it]INFO 04-12 22:29:22 [block_pool.py:378] Successfully reset prefix cache + 3%|▎ | 106/3300 [15:01<9:23:06, 10.58s/it] 2026-04-12 22:29:29,263 | INFO | train_grpo_train | metrics_logged mode=train step=106 + {'loss': 0.1191, 'grad_norm': 20.697811126708984, 'learning_rate': 9.681818181818182e-06, 'num_tokens': 201933.0, 'completions/mean_length': 32.625, 'completions/min_length': 27.0, 'completions/max_length': 49.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 32.625, 'completions/min_terminated_length': 27.0, 'completions/max_terminated_length': 49.0, 'rewards/meter/mean': 0.6116139888763428, 'rewards/meter/std': 0.38843685388565063, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 0.875, 'rewards/hard_gate/std': 0.3535533845424652, 'rewards/repeat_soft/mean': 0.9910503625869751, 'rewards/repeat_soft/std': 0.013928886502981186, 'rewards/judge_quality/mean': 0.5149999856948853, 'rewards/judge_quality/std': 0.2676885426044464, 'rewards/total_composite/mean': 0.5906954407691956, 'rewards/total_composite/std': 0.3208555281162262, 'reward': 0.5906954407691956, 'reward_std': 0.3208554983139038, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.20241546630859375, 'sampling/sampling_logp_difference/max': 1.8645049333572388, 'sampling/importance_sampling_ratio/min': 0.15497291088104248, 'sampling/importance_sampling_ratio/mean': 0.982658863067627, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 0.9796655997633934, 'clip_ratio/low_mean': 0.07287175673991442, 'clip_ratio/low_min': 0.07287175673991442, 'clip_ratio/high_mean': 0.08723545260727406, 'clip_ratio/high_max': 0.08723545260727406, 'clip_ratio/region_mean': 0.16010720934718847, 'reward_total_mean': 0.5906954407691956, 'reward_meter_mean': 0.6116139888763428, 'reward_meter_std': 0.38843685388565063, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 0.875, 'reward_hard_gate_std': 0.3535533845424652, 'reward_repeat_soft_mean': 0.9910503625869751, 'reward_repeat_soft_std': 0.013928886502981186, 'reward_judge_quality_mean': 0.5149999856948853, 'reward_judge_quality_std': 0.2676885426044464, 'reward_total_composite_mean': 0.5906954407691956, 'reward_total_composite_std': 0.3208555281162262, 'epoch': 0.01} + 3%|▎ | 106/3300 [15:01<9:23:06, 10.58s/it]INFO 04-12 22:29:29 [block_pool.py:378] Successfully reset prefix cache + 3%|▎ | 107/3300 [15:07<8:12:50, 9.26s/it]2026-04-12 22:29:35,452 | INFO | train_grpo_train | metrics_logged mode=train step=107 + {'loss': 0.0329, 'grad_norm': 10.300890922546387, 'learning_rate': 9.67878787878788e-06, 'num_tokens': 203724.0, 'completions/mean_length': 60.875, 'completions/min_length': 55.0, 'completions/max_length': 68.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 60.875, 'completions/min_terminated_length': 55.0, 'completions/max_terminated_length': 68.0, 'rewards/meter/mean': 0.6681768894195557, 'rewards/meter/std': 0.33801743388175964, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9916098713874817, 'rewards/repeat_soft/std': 0.010880302637815475, 'rewards/judge_quality/mean': 0.5275000333786011, 'rewards/judge_quality/std': 0.2499571591615677, 'rewards/total_composite/mean': 0.7080905437469482, 'rewards/total_composite/std': 0.16318832337856293, 'reward': 0.7080905437469482, 'reward_std': 0.16318833827972412, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.18917357921600342, 'sampling/sampling_logp_difference/max': 1.667069435119629, 'sampling/importance_sampling_ratio/min': 0.18879954516887665, 'sampling/importance_sampling_ratio/mean': 0.9972324371337891, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.1798932030797005, 'clip_ratio/low_mean': 0.0705209244042635, 'clip_ratio/low_min': 0.0705209244042635, 'clip_ratio/high_mean': 0.0801126342266798, 'clip_ratio/high_max': 0.0801126342266798, 'clip_ratio/region_mean': 0.1506335586309433, 'reward_total_mean': 0.7080905437469482, 'reward_meter_mean': 0.6681768894195557, 'reward_meter_std': 0.33801743388175964, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9916098713874817, 'reward_repeat_soft_std': 0.010880302637815475, 'reward_judge_quality_mean': 0.5275000333786011, 'reward_judge_quality_std': 0.2499571591615677, 'reward_total_composite_mean': 0.7080905437469482, 'reward_total_composite_std': 0.16318832337856293, 'epoch': 0.01} + 3%|▎ | 107/3300 [15:07<8:12:50, 9.26s/it]INFO 04-12 22:29:35 [block_pool.py:378] Successfully reset prefix cache + 3%|▎ | 108/3300 [15:17<8:14:05, 9.29s/it]2026-04-12 22:29:44,800 | INFO | train_grpo_train | metrics_logged mode=train step=108 + {'loss': -0.0065, 'grad_norm': 25.379682540893555, 'learning_rate': 9.675757575757577e-06, 'num_tokens': 205061.0, 'completions/mean_length': 32.125, 'completions/min_length': 23.0, 'completions/max_length': 43.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 32.125, 'completions/min_terminated_length': 23.0, 'completions/max_terminated_length': 43.0, 'rewards/meter/mean': 0.28980332612991333, 'rewards/meter/std': 0.2669162452220917, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9977822303771973, 'rewards/repeat_soft/std': 0.0052426340989768505, 'rewards/judge_quality/mean': 0.7400000095367432, 'rewards/judge_quality/std': 0.24859607219696045, 'rewards/total_composite/mean': 0.6021897196769714, 'rewards/total_composite/std': 0.12454845756292343, 'reward': 0.6021897196769714, 'reward_std': 0.12454845756292343, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.24968045949935913, 'sampling/sampling_logp_difference/max': 2.7263078689575195, 'sampling/importance_sampling_ratio/min': 0.06546053290367126, 'sampling/importance_sampling_ratio/mean': 1.0117802619934082, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.7245535105466843, 'clip_ratio/low_mean': 0.1483722161501646, 'clip_ratio/low_min': 0.1483722161501646, 'clip_ratio/high_mean': 0.06614910159260035, 'clip_ratio/high_max': 0.06614910159260035, 'clip_ratio/region_mean': 0.21452131774276495, 'reward_total_mean': 0.6021897196769714, 'reward_meter_mean': 0.28980332612991333, 'reward_meter_std': 0.2669162452220917, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9977822303771973, 'reward_repeat_soft_std': 0.0052426340989768505, 'reward_judge_quality_mean': 0.7400000095367432, 'reward_judge_quality_std': 0.24859607219696045, 'reward_total_composite_mean': 0.6021897196769714, 'reward_total_composite_std': 0.12454845756292343, 'epoch': 0.01} + 3%|▎ | 108/3300 [15:17<8:14:05, 9.29s/it]INFO 04-12 22:29:45 [block_pool.py:378] Successfully reset prefix cache + 3%|▎ | 109/3300 [15:24<7:38:44, 8.63s/it]2026-04-12 22:29:51,882 | INFO | train_grpo_train | metrics_logged mode=train step=109 + {'loss': -0.0321, 'grad_norm': 17.32401466369629, 'learning_rate': 9.672727272727274e-06, 'num_tokens': 206731.0, 'completions/mean_length': 48.75, 'completions/min_length': 35.0, 'completions/max_length': 65.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 48.75, 'completions/min_terminated_length': 35.0, 'completions/max_terminated_length': 65.0, 'rewards/meter/mean': 0.8012364506721497, 'rewards/meter/std': 0.29985368251800537, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 0.875, 'rewards/hard_gate/std': 0.3535533845424652, 'rewards/repeat_soft/mean': 0.9873402118682861, 'rewards/repeat_soft/std': 0.007849778980016708, 'rewards/judge_quality/mean': 0.5487499833106995, 'rewards/judge_quality/std': 0.22937417030334473, 'rewards/total_composite/mean': 0.6714390516281128, 'rewards/total_composite/std': 0.3215502202510834, 'reward': 0.6714390516281128, 'reward_std': 0.3215502202510834, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.21072760224342346, 'sampling/sampling_logp_difference/max': 1.7081098556518555, 'sampling/importance_sampling_ratio/min': 0.18120796978473663, 'sampling/importance_sampling_ratio/mean': 1.0020984411239624, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.047327294945717, 'clip_ratio/low_mean': 0.062176925130188465, 'clip_ratio/low_min': 0.062176925130188465, 'clip_ratio/high_mean': 0.15053118951618671, 'clip_ratio/high_max': 0.15053118951618671, 'clip_ratio/region_mean': 0.21270811464637518, 'reward_total_mean': 0.6714390516281128, 'reward_meter_mean': 0.8012364506721497, 'reward_meter_std': 0.29985368251800537, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 0.875, 'reward_hard_gate_std': 0.3535533845424652, 'reward_repeat_soft_mean': 0.9873402118682861, 'reward_repeat_soft_std': 0.007849778980016708, 'reward_judge_quality_mean': 0.5487499833106995, 'reward_judge_quality_std': 0.22937417030334473, 'reward_total_composite_mean': 0.6714390516281128, 'reward_total_composite_std': 0.3215502202510834, 'epoch': 0.01} + 3%|▎ | 109/3300 [15:24<7:38:44, 8.63s/it]INFO 04-12 22:29:52 [block_pool.py:378] Successfully reset prefix cache + 3%|▎ | 110/3300 [15:30<7:06:05, 8.01s/it]2026-04-12 22:29:58,471 | INFO | train_grpo_train | metrics_logged mode=train step=110 + {'loss': 0.0301, 'grad_norm': 12.794631004333496, 'learning_rate': 9.66969696969697e-06, 'num_tokens': 208570.0, 'completions/mean_length': 60.875, 'completions/min_length': 35.0, 'completions/max_length': 72.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 60.875, 'completions/min_terminated_length': 35.0, 'completions/max_terminated_length': 72.0, 'rewards/meter/mean': 0.881321907043457, 'rewards/meter/std': 0.29402533173561096, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9829816818237305, 'rewards/repeat_soft/std': 0.019399171695113182, 'rewards/judge_quality/mean': 0.4350000023841858, 'rewards/judge_quality/std': 0.01603567600250244, 'rewards/total_composite/mean': 0.7753930687904358, 'rewards/total_composite/std': 0.13116681575775146, 'reward': 0.7753930687904358, 'reward_std': 0.13116681575775146, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.1848183274269104, 'sampling/sampling_logp_difference/max': 1.5729122161865234, 'sampling/importance_sampling_ratio/min': 0.20744018256664276, 'sampling/importance_sampling_ratio/mean': 1.0221983194351196, 'sampling/importance_sampling_ratio/max': 1.9403345584869385, 'entropy': 1.8935251384973526, 'clip_ratio/low_mean': 0.02330508455634117, 'clip_ratio/low_min': 0.02330508455634117, 'clip_ratio/high_mean': 0.1399045344442129, 'clip_ratio/high_max': 0.1399045344442129, 'clip_ratio/region_mean': 0.16320961900055408, 'reward_total_mean': 0.7753930687904358, 'reward_meter_mean': 0.881321907043457, 'reward_meter_std': 0.29402533173561096, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9829816818237305, 'reward_repeat_soft_std': 0.019399171695113182, 'reward_judge_quality_mean': 0.4350000023841858, 'reward_judge_quality_std': 0.01603567600250244, 'reward_total_composite_mean': 0.7753930687904358, 'reward_total_composite_std': 0.13116681575775146, 'epoch': 0.01} + 3%|▎ | 110/3300 [15:30<7:06:05, 8.01s/it]INFO 04-12 22:29:58 [block_pool.py:378] Successfully reset prefix cache + 3%|▎ | 111/3300 [15:38<6:59:08, 7.89s/it]2026-04-12 22:30:06,062 | INFO | train_grpo_train | metrics_logged mode=train step=111 + {'loss': -0.049, 'grad_norm': 8.582052230834961, 'learning_rate': 9.666666666666667e-06, 'num_tokens': 210836.0, 'completions/mean_length': 119.25, 'completions/min_length': 87.0, 'completions/max_length': 137.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 119.25, 'completions/min_terminated_length': 87.0, 'completions/max_terminated_length': 137.0, 'rewards/meter/mean': 0.6979060173034668, 'rewards/meter/std': 0.3188508450984955, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9957622289657593, 'rewards/repeat_soft/std': 0.005219956859946251, 'rewards/judge_quality/mean': 0.42374998331069946, 'rewards/judge_quality/std': 0.1524970829486847, 'rewards/total_composite/mean': 0.6907588839530945, 'rewards/total_composite/std': 0.15496724843978882, 'reward': 0.6907588839530945, 'reward_std': 0.15496723353862762, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.19657625257968903, 'sampling/sampling_logp_difference/max': 1.2601728439331055, 'sampling/importance_sampling_ratio/min': 0.28360500931739807, 'sampling/importance_sampling_ratio/mean': 1.035998821258545, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.517204165458679, 'clip_ratio/low_mean': 0.09810720011591911, 'clip_ratio/low_min': 0.09810720011591911, 'clip_ratio/high_mean': 0.09997841343283653, 'clip_ratio/high_max': 0.09997841343283653, 'clip_ratio/region_mean': 0.19808561354875565, 'reward_total_mean': 0.6907588839530945, 'reward_meter_mean': 0.6979060173034668, 'reward_meter_std': 0.3188508450984955, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9957622289657593, 'reward_repeat_soft_std': 0.005219956859946251, 'reward_judge_quality_mean': 0.42374998331069946, 'reward_judge_quality_std': 0.1524970829486847, 'reward_total_composite_mean': 0.6907588839530945, 'reward_total_composite_std': 0.15496724843978882, 'epoch': 0.01} + 3%|▎ | 111/3300 [15:38<6:59:08, 7.89s/it]INFO 04-12 22:30:06 [block_pool.py:378] Successfully reset prefix cache + 3%|▎ | 112/3300 [15:44<6:29:46, 7.34s/it]2026-04-12 22:30:12,129 | INFO | train_grpo_train | metrics_logged mode=train step=112 + {'loss': 0.0122, 'grad_norm': 20.0980224609375, 'learning_rate': 9.663636363636364e-06, 'num_tokens': 212352.0, 'completions/mean_length': 24.5, 'completions/min_length': 17.0, 'completions/max_length': 31.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 24.5, 'completions/min_terminated_length': 17.0, 'completions/max_terminated_length': 31.0, 'rewards/meter/mean': 0.5157471299171448, 'rewards/meter/std': 0.40246090292930603, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9624999761581421, 'rewards/repeat_soft/std': 0.0, 'rewards/judge_quality/mean': 0.5137500166893005, 'rewards/judge_quality/std': 0.26462578773498535, 'rewards/total_composite/mean': 0.6324611902236938, 'rewards/total_composite/std': 0.1423051655292511, 'reward': 0.6324611902236938, 'reward_std': 0.1423051506280899, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.1752002090215683, 'sampling/sampling_logp_difference/max': 1.0431177616119385, 'sampling/importance_sampling_ratio/min': 0.35235440731048584, 'sampling/importance_sampling_ratio/mean': 1.0118403434753418, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.3280086815357208, 'clip_ratio/low_mean': 0.09726544190198183, 'clip_ratio/low_min': 0.09726544190198183, 'clip_ratio/high_mean': 0.05161340953782201, 'clip_ratio/high_max': 0.05161340953782201, 'clip_ratio/region_mean': 0.14887885143980384, 'reward_total_mean': 0.6324611902236938, 'reward_meter_mean': 0.5157471299171448, 'reward_meter_std': 0.40246090292930603, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9624999761581421, 'reward_repeat_soft_std': 0.0, 'reward_judge_quality_mean': 0.5137500166893005, 'reward_judge_quality_std': 0.26462578773498535, 'reward_total_composite_mean': 0.6324611902236938, 'reward_total_composite_std': 0.1423051655292511, 'epoch': 0.01} + 3%|▎ | 112/3300 [15:44<6:29:46, 7.34s/it]INFO 04-12 22:30:12 [block_pool.py:378] Successfully reset prefix cache + 3%|▎ | 113/3300 [15:50<6:14:48, 7.06s/it]2026-04-12 22:30:18,514 | INFO | train_grpo_train | metrics_logged mode=train step=113 + {'loss': 0.0815, 'grad_norm': 12.603521347045898, 'learning_rate': 9.660606060606061e-06, 'num_tokens': 214242.0, 'completions/mean_length': 80.25, 'completions/min_length': 72.0, 'completions/max_length': 95.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 80.25, 'completions/min_terminated_length': 72.0, 'completions/max_terminated_length': 95.0, 'rewards/meter/mean': 0.35283008217811584, 'rewards/meter/std': 0.32658228278160095, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9970334768295288, 'rewards/repeat_soft/std': 0.004829896613955498, 'rewards/judge_quality/mean': 0.6737500429153442, 'rewards/judge_quality/std': 0.263435423374176, 'rewards/total_composite/mean': 0.6106019020080566, 'rewards/total_composite/std': 0.17289768159389496, 'reward': 0.6106019020080566, 'reward_std': 0.17289769649505615, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.18053029477596283, 'sampling/sampling_logp_difference/max': 2.1333065032958984, 'sampling/importance_sampling_ratio/min': 0.1184450089931488, 'sampling/importance_sampling_ratio/mean': 1.0171382427215576, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.4742966517806053, 'clip_ratio/low_mean': 0.07181645464152098, 'clip_ratio/low_min': 0.07181645464152098, 'clip_ratio/high_mean': 0.08113026432693005, 'clip_ratio/high_max': 0.08113026432693005, 'clip_ratio/region_mean': 0.15294671896845102, 'reward_total_mean': 0.6106019020080566, 'reward_meter_mean': 0.35283008217811584, 'reward_meter_std': 0.32658228278160095, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9970334768295288, 'reward_repeat_soft_std': 0.004829896613955498, 'reward_judge_quality_mean': 0.6737500429153442, 'reward_judge_quality_std': 0.263435423374176, 'reward_total_composite_mean': 0.6106019020080566, 'reward_total_composite_std': 0.17289768159389496, 'epoch': 0.01} + 3%|▎ | 113/3300 [15:50<6:14:48, 7.06s/it]INFO 04-12 22:30:18 [block_pool.py:378] Successfully reset prefix cache + 3%|▎ | 114/3300 [15:57<6:10:49, 6.98s/it]2026-04-12 22:30:25,327 | INFO | train_grpo_train | metrics_logged mode=train step=114 + {'loss': 0.0679, 'grad_norm': 13.874102592468262, 'learning_rate': 9.657575757575758e-06, 'num_tokens': 215968.0, 'completions/mean_length': 61.75, 'completions/min_length': 44.0, 'completions/max_length': 74.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 61.75, 'completions/min_terminated_length': 44.0, 'completions/max_terminated_length': 74.0, 'rewards/meter/mean': 0.555526614189148, 'rewards/meter/std': 0.47907620668411255, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.994547963142395, 'rewards/repeat_soft/std': 0.013426722027361393, 'rewards/judge_quality/mean': 0.35999998450279236, 'rewards/judge_quality/std': 0.09165150672197342, 'rewards/total_composite/mean': 0.607441782951355, 'rewards/total_composite/std': 0.2246592491865158, 'reward': 0.607441782951355, 'reward_std': 0.2246592491865158, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.19817504286766052, 'sampling/sampling_logp_difference/max': 1.2710886001586914, 'sampling/importance_sampling_ratio/min': 0.2805260717868805, 'sampling/importance_sampling_ratio/mean': 1.0383985042572021, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.0831169039011, 'clip_ratio/low_mean': 0.07543539721518755, 'clip_ratio/low_min': 0.07543539721518755, 'clip_ratio/high_mean': 0.10498625971376896, 'clip_ratio/high_max': 0.10498625971376896, 'clip_ratio/region_mean': 0.1804216569289565, 'reward_total_mean': 0.607441782951355, 'reward_meter_mean': 0.555526614189148, 'reward_meter_std': 0.47907620668411255, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.994547963142395, 'reward_repeat_soft_std': 0.013426722027361393, 'reward_judge_quality_mean': 0.35999998450279236, 'reward_judge_quality_std': 0.09165150672197342, 'reward_total_composite_mean': 0.607441782951355, 'reward_total_composite_std': 0.2246592491865158, 'epoch': 0.01} + 3%|▎ | 114/3300 [15:57<6:10:49, 6.98s/it]INFO 04-12 22:30:25 [block_pool.py:378] Successfully reset prefix cache + 3%|▎ | 115/3300 [16:05<6:25:26, 7.26s/it]2026-04-12 22:30:33,236 | INFO | train_grpo_train | metrics_logged mode=train step=115 + {'loss': -0.012, 'grad_norm': 9.302298545837402, 'learning_rate': 9.654545454545456e-06, 'num_tokens': 218404.0, 'completions/mean_length': 116.5, 'completions/min_length': 95.0, 'completions/max_length': 131.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 116.5, 'completions/min_terminated_length': 95.0, 'completions/max_terminated_length': 131.0, 'rewards/meter/mean': 0.5344254970550537, 'rewards/meter/std': 0.3820936977863312, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.99835604429245, 'rewards/repeat_soft/std': 0.0023890030570328236, 'rewards/judge_quality/mean': 0.47749996185302734, 'rewards/judge_quality/std': 0.16263456642627716, 'rewards/total_composite/mean': 0.6335770487785339, 'rewards/total_composite/std': 0.19988654553890228, 'reward': 0.6335770487785339, 'reward_std': 0.19988654553890228, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.21457158029079437, 'sampling/sampling_logp_difference/max': 2.245924949645996, 'sampling/importance_sampling_ratio/min': 0.10582961142063141, 'sampling/importance_sampling_ratio/mean': 1.0353453159332275, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.466738671064377, 'clip_ratio/low_mean': 0.09431399963796139, 'clip_ratio/low_min': 0.09431399963796139, 'clip_ratio/high_mean': 0.10586324706673622, 'clip_ratio/high_max': 0.10586324706673622, 'clip_ratio/region_mean': 0.2001772467046976, 'reward_total_mean': 0.6335770487785339, 'reward_meter_mean': 0.5344254970550537, 'reward_meter_std': 0.3820936977863312, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.99835604429245, 'reward_repeat_soft_std': 0.0023890030570328236, 'reward_judge_quality_mean': 0.47749996185302734, 'reward_judge_quality_std': 0.16263456642627716, 'reward_total_composite_mean': 0.6335770487785339, 'reward_total_composite_std': 0.19988654553890228, 'epoch': 0.01} + 3%|▎ | 115/3300 [16:05<6:25:26, 7.26s/it]INFO 04-12 22:30:33 [block_pool.py:378] Successfully reset prefix cache + 4%|▎ | 116/3300 [16:11<6:04:20, 6.87s/it]2026-04-12 22:30:39,178 | INFO | train_grpo_train | metrics_logged mode=train step=116 + {'loss': 0.0721, 'grad_norm': 14.912810325622559, 'learning_rate': 9.651515151515153e-06, 'num_tokens': 220027.0, 'completions/mean_length': 47.875, 'completions/min_length': 41.0, 'completions/max_length': 57.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 47.875, 'completions/min_terminated_length': 41.0, 'completions/max_terminated_length': 57.0, 'rewards/meter/mean': 0.5695972442626953, 'rewards/meter/std': 0.41784724593162537, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9951172471046448, 'rewards/repeat_soft/std': 0.007368812803179026, 'rewards/judge_quality/mean': 0.6987500190734863, 'rewards/judge_quality/std': 0.24474114179611206, 'rewards/total_composite/mean': 0.7154554128646851, 'rewards/total_composite/std': 0.19057126343250275, 'reward': 0.7154554128646851, 'reward_std': 0.19057126343250275, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.17878463864326477, 'sampling/sampling_logp_difference/max': 1.5432236194610596, 'sampling/importance_sampling_ratio/min': 0.2136911302804947, 'sampling/importance_sampling_ratio/mean': 1.0183104276657104, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.463243305683136, 'clip_ratio/low_mean': 0.08487668633460999, 'clip_ratio/low_min': 0.08487668633460999, 'clip_ratio/high_mean': 0.1055582882836461, 'clip_ratio/high_max': 0.1055582882836461, 'clip_ratio/region_mean': 0.1904349746182561, 'reward_total_mean': 0.7154554128646851, 'reward_meter_mean': 0.5695972442626953, 'reward_meter_std': 0.41784724593162537, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9951172471046448, 'reward_repeat_soft_std': 0.007368812803179026, 'reward_judge_quality_mean': 0.6987500190734863, 'reward_judge_quality_std': 0.24474114179611206, 'reward_total_composite_mean': 0.7154554128646851, 'reward_total_composite_std': 0.19057126343250275, 'epoch': 0.01} + 4%|▎ | 116/3300 [16:11<6:04:20, 6.87s/it]INFO 04-12 22:30:39 [block_pool.py:378] Successfully reset prefix cache + 4%|▎ | 117/3300 [16:17<5:52:43, 6.65s/it]2026-04-12 22:30:45,323 | INFO | train_grpo_train | metrics_logged mode=train step=117 + {'loss': 0.013, 'grad_norm': 19.91445541381836, 'learning_rate': 9.648484848484849e-06, 'num_tokens': 221624.0, 'completions/mean_length': 48.625, 'completions/min_length': 36.0, 'completions/max_length': 67.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 48.625, 'completions/min_terminated_length': 36.0, 'completions/max_terminated_length': 67.0, 'rewards/meter/mean': 0.9055629372596741, 'rewards/meter/std': 0.1779366135597229, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9891988039016724, 'rewards/repeat_soft/std': 0.002124833408743143, 'rewards/judge_quality/mean': 0.8650000095367432, 'rewards/judge_quality/std': 0.09055384993553162, 'rewards/total_composite/mean': 0.9159232378005981, 'rewards/total_composite/std': 0.1001143530011177, 'reward': 0.9159232378005981, 'reward_std': 0.10011434555053711, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.08398713171482086, 'sampling/sampling_logp_difference/max': 1.432332992553711, 'sampling/importance_sampling_ratio/min': 0.23875127732753754, 'sampling/importance_sampling_ratio/mean': 0.9926836490631104, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 0.28090511448681355, 'clip_ratio/low_mean': 0.022110641933977604, 'clip_ratio/low_min': 0.022110641933977604, 'clip_ratio/high_mean': 0.039251322858035564, 'clip_ratio/high_max': 0.039251322858035564, 'clip_ratio/region_mean': 0.06136196479201317, 'reward_total_mean': 0.9159232378005981, 'reward_meter_mean': 0.9055629372596741, 'reward_meter_std': 0.1779366135597229, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9891988039016724, 'reward_repeat_soft_std': 0.002124833408743143, 'reward_judge_quality_mean': 0.8650000095367432, 'reward_judge_quality_std': 0.09055384993553162, 'reward_total_composite_mean': 0.9159232378005981, 'reward_total_composite_std': 0.1001143530011177, 'epoch': 0.01} + 4%|▎ | 117/3300 [16:17<5:52:43, 6.65s/it]INFO 04-12 22:30:45 [block_pool.py:378] Successfully reset prefix cache + 4%|▎ | 118/3300 [16:25<6:15:13, 7.08s/it]2026-04-12 22:30:53,396 | INFO | train_grpo_train | metrics_logged mode=train step=118 + {'loss': -0.0515, 'grad_norm': 7.082086086273193, 'learning_rate': 9.645454545454548e-06, 'num_tokens': 224495.0, 'completions/mean_length': 158.875, 'completions/min_length': 130.0, 'completions/max_length': 180.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 158.875, 'completions/min_terminated_length': 130.0, 'completions/max_terminated_length': 180.0, 'rewards/meter/mean': 0.7066109776496887, 'rewards/meter/std': 0.28142380714416504, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 0.875, 'rewards/hard_gate/std': 0.3535533845424652, 'rewards/repeat_soft/mean': 0.9959703087806702, 'rewards/repeat_soft/std': 0.0038161210250109434, 'rewards/judge_quality/mean': 0.3987500071525574, 'rewards/judge_quality/std': 0.06010407209396362, 'rewards/total_composite/mean': 0.631685733795166, 'rewards/total_composite/std': 0.27332279086112976, 'reward': 0.631685733795166, 'reward_std': 0.27332279086112976, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.20089176297187805, 'sampling/sampling_logp_difference/max': 1.853123664855957, 'sampling/importance_sampling_ratio/min': 0.1567467898130417, 'sampling/importance_sampling_ratio/mean': 1.0425838232040405, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.4911645650863647, 'clip_ratio/low_mean': 0.07768536917865276, 'clip_ratio/low_min': 0.07768536917865276, 'clip_ratio/high_mean': 0.12430565804243088, 'clip_ratio/high_max': 0.12430565804243088, 'clip_ratio/region_mean': 0.20199102722108364, 'reward_total_mean': 0.631685733795166, 'reward_meter_mean': 0.7066109776496887, 'reward_meter_std': 0.28142380714416504, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 0.875, 'reward_hard_gate_std': 0.3535533845424652, 'reward_repeat_soft_mean': 0.9959703087806702, 'reward_repeat_soft_std': 0.0038161210250109434, 'reward_judge_quality_mean': 0.3987500071525574, 'reward_judge_quality_std': 0.06010407209396362, 'reward_total_composite_mean': 0.631685733795166, 'reward_total_composite_std': 0.27332279086112976, 'epoch': 0.01} + 4%|▎ | 118/3300 [16:25<6:15:13, 7.08s/it]INFO 04-12 22:30:53 [block_pool.py:378] Successfully reset prefix cache + 4%|▎ | 119/3300 [16:32<6:10:15, 6.98s/it]2026-04-12 22:31:00,163 | INFO | train_grpo_train | metrics_logged mode=train step=119 + {'loss': -0.0428, 'grad_norm': 12.962621688842773, 'learning_rate': 9.642424242424243e-06, 'num_tokens': 226327.0, 'completions/mean_length': 63.0, 'completions/min_length': 44.0, 'completions/max_length': 72.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 63.0, 'completions/min_terminated_length': 44.0, 'completions/max_terminated_length': 72.0, 'rewards/meter/mean': 0.7072560787200928, 'rewards/meter/std': 0.3508759140968323, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 0.875, 'rewards/hard_gate/std': 0.3535533845424652, 'rewards/repeat_soft/mean': 0.996657133102417, 'rewards/repeat_soft/std': 0.0072163003496825695, 'rewards/judge_quality/mean': 0.45625001192092896, 'rewards/judge_quality/std': 0.21185828745365143, 'rewards/total_composite/mean': 0.6302478313446045, 'rewards/total_composite/std': 0.31918975710868835, 'reward': 0.6302478313446045, 'reward_std': 0.31918975710868835, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.1847275048494339, 'sampling/sampling_logp_difference/max': 1.2999019622802734, 'sampling/importance_sampling_ratio/min': 0.2725585103034973, 'sampling/importance_sampling_ratio/mean': 1.0211437940597534, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.0329722613096237, 'clip_ratio/low_mean': 0.05912642180919647, 'clip_ratio/low_min': 0.05912642180919647, 'clip_ratio/high_mean': 0.12136753089725971, 'clip_ratio/high_max': 0.12136753089725971, 'clip_ratio/region_mean': 0.18049395270645618, 'reward_total_mean': 0.6302478313446045, 'reward_meter_mean': 0.7072560787200928, 'reward_meter_std': 0.3508759140968323, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 0.875, 'reward_hard_gate_std': 0.3535533845424652, 'reward_repeat_soft_mean': 0.996657133102417, 'reward_repeat_soft_std': 0.0072163003496825695, 'reward_judge_quality_mean': 0.45625001192092896, 'reward_judge_quality_std': 0.21185828745365143, 'reward_total_composite_mean': 0.6302478313446045, 'reward_total_composite_std': 0.31918975710868835, 'epoch': 0.01} + 4%|▎ | 119/3300 [16:32<6:10:15, 6.98s/it]INFO 04-12 22:31:00 [block_pool.py:378] Successfully reset prefix cache + 4%|▎ | 120/3300 [16:44<7:26:19, 8.42s/it]2026-04-12 22:31:11,938 | INFO | train_grpo_train | metrics_logged mode=train step=120 + {'loss': -0.1991, 'grad_norm': 3.504687547683716, 'learning_rate': 9.63939393939394e-06, 'num_tokens': 228691.0, 'completions/mean_length': 165.5, 'completions/min_length': 94.0, 'completions/max_length': 512.0, 'completions/clipped_ratio': 0.125, 'completions/mean_terminated_length': 116.00000762939453, 'completions/min_terminated_length': 94.0, 'completions/max_terminated_length': 131.0, 'rewards/meter/mean': 0.2696044445037842, 'rewards/meter/std': 0.2580581307411194, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 0.875, 'rewards/hard_gate/std': 0.3535533845424652, 'rewards/repeat_soft/mean': 0.9958187341690063, 'rewards/repeat_soft/std': 0.003901544027030468, 'rewards/judge_quality/mean': 0.398749977350235, 'rewards/judge_quality/std': 0.17125065624713898, 'rewards/total_composite/mean': 0.4501357078552246, 'rewards/total_composite/std': 0.20900095999240875, 'reward': 0.4501357078552246, 'reward_std': 0.20900094509124756, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.1954592615365982, 'sampling/sampling_logp_difference/max': 1.9527530670166016, 'sampling/importance_sampling_ratio/min': 0.14188291132450104, 'sampling/importance_sampling_ratio/mean': 1.022305965423584, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.7254022061824799, 'clip_ratio/low_mean': 0.04175109788775444, 'clip_ratio/low_min': 0.04175109788775444, 'clip_ratio/high_mean': 0.13687382265925407, 'clip_ratio/high_max': 0.13687382265925407, 'clip_ratio/region_mean': 0.17862492054700851, 'reward_total_mean': 0.4501357078552246, 'reward_meter_mean': 0.2696044445037842, 'reward_meter_std': 0.2580581307411194, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 0.875, 'reward_hard_gate_std': 0.3535533845424652, 'reward_repeat_soft_mean': 0.9958187341690063, 'reward_repeat_soft_std': 0.003901544027030468, 'reward_judge_quality_mean': 0.398749977350235, 'reward_judge_quality_std': 0.17125065624713898, 'reward_total_composite_mean': 0.4501357078552246, 'reward_total_composite_std': 0.20900095999240875, 'epoch': 0.01} + 4%|▎ | 120/3300 [16:44<7:26:19, 8.42s/it]INFO 04-12 22:31:12 [block_pool.py:378] Successfully reset prefix cache + 4%|▎ | 121/3300 [16:52<7:21:32, 8.33s/it]2026-04-12 22:31:20,069 | INFO | train_grpo_train | metrics_logged mode=train step=121 + {'loss': 0.0322, 'grad_norm': 8.148179054260254, 'learning_rate': 9.636363636363638e-06, 'num_tokens': 231502.0, 'completions/mean_length': 151.375, 'completions/min_length': 122.0, 'completions/max_length': 171.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 151.375, 'completions/min_terminated_length': 122.0, 'completions/max_terminated_length': 171.0, 'rewards/meter/mean': 0.46986302733421326, 'rewards/meter/std': 0.25895586609840393, 'rewards/count_adherence/mean': 0.949999988079071, 'rewards/count_adherence/std': 0.09258200973272324, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9900440573692322, 'rewards/repeat_soft/std': 0.007098556496202946, 'rewards/judge_quality/mean': 0.4362500011920929, 'rewards/judge_quality/std': 0.12916629016399384, 'rewards/total_composite/mean': 0.5838177800178528, 'rewards/total_composite/std': 0.12632137537002563, 'reward': 0.5838177800178528, 'reward_std': 0.12632139027118683, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.20250819623470306, 'sampling/sampling_logp_difference/max': 1.9475808143615723, 'sampling/importance_sampling_ratio/min': 0.14261867105960846, 'sampling/importance_sampling_ratio/mean': 1.0346516370773315, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.129692703485489, 'clip_ratio/low_mean': 0.10343983769416809, 'clip_ratio/low_min': 0.10343983769416809, 'clip_ratio/high_mean': 0.10095386765897274, 'clip_ratio/high_max': 0.10095386765897274, 'clip_ratio/region_mean': 0.20439370535314083, 'reward_total_mean': 0.5838177800178528, 'reward_meter_mean': 0.46986302733421326, 'reward_meter_std': 0.25895586609840393, 'reward_count_adherence_mean': 0.949999988079071, 'reward_count_adherence_std': 0.09258200973272324, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9900440573692322, 'reward_repeat_soft_std': 0.007098556496202946, 'reward_judge_quality_mean': 0.4362500011920929, 'reward_judge_quality_std': 0.12916629016399384, 'reward_total_composite_mean': 0.5838177800178528, 'reward_total_composite_std': 0.12632137537002563, 'epoch': 0.01} + 4%|▎ | 121/3300 [16:52<7:21:32, 8.33s/it]INFO 04-12 22:31:20 [block_pool.py:378] Successfully reset prefix cache + 4%|▎ | 122/3300 [16:59<6:51:58, 7.78s/it]2026-04-12 22:31:26,550 | INFO | train_grpo_train | metrics_logged mode=train step=122 + {'loss': -0.0128, 'grad_norm': 10.772955894470215, 'learning_rate': 9.633333333333335e-06, 'num_tokens': 233265.0, 'completions/mean_length': 62.375, 'completions/min_length': 49.0, 'completions/max_length': 77.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 62.375, 'completions/min_terminated_length': 49.0, 'completions/max_terminated_length': 77.0, 'rewards/meter/mean': 0.840729832649231, 'rewards/meter/std': 0.21353866159915924, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 0.75, 'rewards/hard_gate/std': 0.4629100561141968, 'rewards/repeat_soft/mean': 0.9988186359405518, 'rewards/repeat_soft/std': 0.0025750070344656706, 'rewards/judge_quality/mean': 0.5175000429153442, 'rewards/judge_quality/std': 0.12498573213815689, 'rewards/total_composite/mean': 0.5670230984687805, 'rewards/total_composite/std': 0.36396607756614685, 'reward': 0.5670230984687805, 'reward_std': 0.36396604776382446, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.21837672591209412, 'sampling/sampling_logp_difference/max': 1.421654224395752, 'sampling/importance_sampling_ratio/min': 0.24131451547145844, 'sampling/importance_sampling_ratio/mean': 1.0179781913757324, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.149739980697632, 'clip_ratio/low_mean': 0.046270232647657394, 'clip_ratio/low_min': 0.046270232647657394, 'clip_ratio/high_mean': 0.15077135432511568, 'clip_ratio/high_max': 0.15077135432511568, 'clip_ratio/region_mean': 0.19704158697277308, 'reward_total_mean': 0.5670230984687805, 'reward_meter_mean': 0.840729832649231, 'reward_meter_std': 0.21353866159915924, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 0.75, 'reward_hard_gate_std': 0.4629100561141968, 'reward_repeat_soft_mean': 0.9988186359405518, 'reward_repeat_soft_std': 0.0025750070344656706, 'reward_judge_quality_mean': 0.5175000429153442, 'reward_judge_quality_std': 0.12498573213815689, 'reward_total_composite_mean': 0.5670230984687805, 'reward_total_composite_std': 0.36396607756614685, 'epoch': 0.01} + 4%|▎ | 122/3300 [16:59<6:51:58, 7.78s/it]INFO 04-12 22:31:26 [block_pool.py:378] Successfully reset prefix cache + 4%|▎ | 123/3300 [17:05<6:23:50, 7.25s/it]2026-04-12 22:31:32,566 | INFO | train_grpo_train | metrics_logged mode=train step=123 + {'loss': -0.0751, 'grad_norm': 19.844261169433594, 'learning_rate': 9.63030303030303e-06, 'num_tokens': 234744.0, 'completions/mean_length': 36.875, 'completions/min_length': 26.0, 'completions/max_length': 48.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 36.875, 'completions/min_terminated_length': 26.0, 'completions/max_terminated_length': 48.0, 'rewards/meter/mean': 0.154674232006073, 'rewards/meter/std': 0.19714823365211487, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 0.875, 'rewards/hard_gate/std': 0.3535533845424652, 'rewards/repeat_soft/mean': 0.9931535720825195, 'rewards/repeat_soft/std': 0.00823448970913887, 'rewards/judge_quality/mean': 0.6525000333786011, 'rewards/judge_quality/std': 0.2921227812767029, 'rewards/total_composite/mean': 0.47386685013771057, 'rewards/total_composite/std': 0.22027365863323212, 'reward': 0.47386685013771057, 'reward_std': 0.22027365863323212, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.19463254511356354, 'sampling/sampling_logp_difference/max': 1.9416074752807617, 'sampling/importance_sampling_ratio/min': 0.14347313344478607, 'sampling/importance_sampling_ratio/mean': 1.0010665655136108, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.0737264752388, 'clip_ratio/low_mean': 0.0697358213365078, 'clip_ratio/low_min': 0.0697358213365078, 'clip_ratio/high_mean': 0.08342572208493948, 'clip_ratio/high_max': 0.08342572208493948, 'clip_ratio/region_mean': 0.15316154342144728, 'reward_total_mean': 0.47386685013771057, 'reward_meter_mean': 0.154674232006073, 'reward_meter_std': 0.19714823365211487, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 0.875, 'reward_hard_gate_std': 0.3535533845424652, 'reward_repeat_soft_mean': 0.9931535720825195, 'reward_repeat_soft_std': 0.00823448970913887, 'reward_judge_quality_mean': 0.6525000333786011, 'reward_judge_quality_std': 0.2921227812767029, 'reward_total_composite_mean': 0.47386685013771057, 'reward_total_composite_std': 0.22027365863323212, 'epoch': 0.01} + 4%|▎ | 123/3300 [17:05<6:23:50, 7.25s/it]INFO 04-12 22:31:32 [block_pool.py:378] Successfully reset prefix cache + 4%|▍ | 124/3300 [17:11<6:13:03, 7.05s/it]2026-04-12 22:31:39,144 | INFO | train_grpo_train | metrics_logged mode=train step=124 + {'loss': 0.0995, 'grad_norm': 22.603771209716797, 'learning_rate': 9.627272727272728e-06, 'num_tokens': 236144.0, 'completions/mean_length': 26.0, 'completions/min_length': 18.0, 'completions/max_length': 34.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 26.0, 'completions/min_terminated_length': 18.0, 'completions/max_terminated_length': 34.0, 'rewards/meter/mean': 0.7815031409263611, 'rewards/meter/std': 0.372760534286499, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.959294855594635, 'rewards/repeat_soft/std': 0.009065471589565277, 'rewards/judge_quality/mean': 0.49000000953674316, 'rewards/judge_quality/std': 0.1742740124464035, 'rewards/total_composite/mean': 0.7446058988571167, 'rewards/total_composite/std': 0.18518446385860443, 'reward': 0.7446058988571167, 'reward_std': 0.18518446385860443, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.18050359189510345, 'sampling/sampling_logp_difference/max': 1.3536615371704102, 'sampling/importance_sampling_ratio/min': 0.2747851610183716, 'sampling/importance_sampling_ratio/mean': 1.0089120864868164, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.512794516980648, 'clip_ratio/low_mean': 0.03327047452330589, 'clip_ratio/low_min': 0.03327047452330589, 'clip_ratio/high_mean': 0.14881853945553303, 'clip_ratio/high_max': 0.14881853945553303, 'clip_ratio/region_mean': 0.18208901397883892, 'reward_total_mean': 0.7446058988571167, 'reward_meter_mean': 0.7815031409263611, 'reward_meter_std': 0.372760534286499, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.959294855594635, 'reward_repeat_soft_std': 0.009065471589565277, 'reward_judge_quality_mean': 0.49000000953674316, 'reward_judge_quality_std': 0.1742740124464035, 'reward_total_composite_mean': 0.7446058988571167, 'reward_total_composite_std': 0.18518446385860443, 'epoch': 0.01} + 4%|▍ | 124/3300 [17:11<6:13:03, 7.05s/it]INFO 04-12 22:31:39 [block_pool.py:378] Successfully reset prefix cache + 4%|▍ | 125/3300 [17:18<6:12:49, 7.05s/it]2026-04-12 22:31:46,182 | INFO | train_grpo_train | metrics_logged mode=train step=125 + {'loss': 0.0231, 'grad_norm': 18.275697708129883, 'learning_rate': 9.624242424242425e-06, 'num_tokens': 238133.0, 'completions/mean_length': 79.625, 'completions/min_length': 64.0, 'completions/max_length': 94.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 79.625, 'completions/min_terminated_length': 64.0, 'completions/max_terminated_length': 94.0, 'rewards/meter/mean': 0.4361572563648224, 'rewards/meter/std': 0.2553798258304596, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 0.875, 'rewards/hard_gate/std': 0.3535533845424652, 'rewards/repeat_soft/mean': 0.9925416707992554, 'rewards/repeat_soft/std': 0.006211057770997286, 'rewards/judge_quality/mean': 0.4950000047683716, 'rewards/judge_quality/std': 0.13887304067611694, 'rewards/total_composite/mean': 0.5389169454574585, 'rewards/total_composite/std': 0.24676160514354706, 'reward': 0.5389169454574585, 'reward_std': 0.24676157534122467, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.22753405570983887, 'sampling/sampling_logp_difference/max': 2.1534345149993896, 'sampling/importance_sampling_ratio/min': 0.11608477681875229, 'sampling/importance_sampling_ratio/mean': 1.0170170068740845, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.442027322947979, 'clip_ratio/low_mean': 0.05522531643509865, 'clip_ratio/low_min': 0.05522531643509865, 'clip_ratio/high_mean': 0.15537106804549694, 'clip_ratio/high_max': 0.15537106804549694, 'clip_ratio/region_mean': 0.2105963844805956, 'reward_total_mean': 0.5389169454574585, 'reward_meter_mean': 0.4361572563648224, 'reward_meter_std': 0.2553798258304596, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 0.875, 'reward_hard_gate_std': 0.3535533845424652, 'reward_repeat_soft_mean': 0.9925416707992554, 'reward_repeat_soft_std': 0.006211057770997286, 'reward_judge_quality_mean': 0.4950000047683716, 'reward_judge_quality_std': 0.13887304067611694, 'reward_total_composite_mean': 0.5389169454574585, 'reward_total_composite_std': 0.24676160514354706, 'epoch': 0.01} + 4%|▍ | 125/3300 [17:18<6:12:49, 7.05s/it]INFO 04-12 22:31:46 [block_pool.py:378] Successfully reset prefix cache + 4%|▍ | 126/3300 [17:24<5:51:11, 6.64s/it]2026-04-12 22:31:51,873 | INFO | train_grpo_train | metrics_logged mode=train step=126 + {'loss': 0.0736, 'grad_norm': 14.959473609924316, 'learning_rate': 9.621212121212122e-06, 'num_tokens': 239531.0, 'completions/mean_length': 30.75, 'completions/min_length': 26.0, 'completions/max_length': 35.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 30.75, 'completions/min_terminated_length': 26.0, 'completions/max_terminated_length': 35.0, 'rewards/meter/mean': 0.9794123768806458, 'rewards/meter/std': 0.026296310126781464, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9514520168304443, 'rewards/repeat_soft/std': 0.01740238256752491, 'rewards/judge_quality/mean': 0.5400000214576721, 'rewards/judge_quality/std': 0.33393755555152893, 'rewards/total_composite/mean': 0.8478807210922241, 'rewards/total_composite/std': 0.10403254628181458, 'reward': 0.8478807210922241, 'reward_std': 0.10403254628181458, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.1731196641921997, 'sampling/sampling_logp_difference/max': 1.1950368881225586, 'sampling/importance_sampling_ratio/min': 0.3026927709579468, 'sampling/importance_sampling_ratio/mean': 1.0100114345550537, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.6873226910829544, 'clip_ratio/low_mean': 0.1017175568267703, 'clip_ratio/low_min': 0.1017175568267703, 'clip_ratio/high_mean': 0.05982143059372902, 'clip_ratio/high_max': 0.05982143059372902, 'clip_ratio/region_mean': 0.16153898742049932, 'reward_total_mean': 0.8478807210922241, 'reward_meter_mean': 0.9794123768806458, 'reward_meter_std': 0.026296310126781464, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9514520168304443, 'reward_repeat_soft_std': 0.01740238256752491, 'reward_judge_quality_mean': 0.5400000214576721, 'reward_judge_quality_std': 0.33393755555152893, 'reward_total_composite_mean': 0.8478807210922241, 'reward_total_composite_std': 0.10403254628181458, 'epoch': 0.01} + 4%|▍ | 126/3300 [17:24<5:51:11, 6.64s/it]INFO 04-12 22:31:52 [block_pool.py:378] Successfully reset prefix cache + 4%|▍ | 127/3300 [17:30<5:47:13, 6.57s/it]2026-04-12 22:31:58,269 | INFO | train_grpo_train | metrics_logged mode=train step=127 + {'loss': 0.0008, 'grad_norm': 13.659857749938965, 'learning_rate': 9.61818181818182e-06, 'num_tokens': 241182.0, 'completions/mean_length': 60.375, 'completions/min_length': 53.0, 'completions/max_length': 66.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 60.375, 'completions/min_terminated_length': 53.0, 'completions/max_terminated_length': 66.0, 'rewards/meter/mean': 0.6842314004898071, 'rewards/meter/std': 0.4053438603878021, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 1.0, 'rewards/repeat_soft/std': 0.0, 'rewards/judge_quality/mean': 0.40625, 'rewards/judge_quality/std': 0.0645727664232254, 'rewards/total_composite/mean': 0.6797791123390198, 'rewards/total_composite/std': 0.17921513319015503, 'reward': 0.6797791123390198, 'reward_std': 0.17921513319015503, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.20752720534801483, 'sampling/sampling_logp_difference/max': 1.3170757293701172, 'sampling/importance_sampling_ratio/min': 0.2679176330566406, 'sampling/importance_sampling_ratio/mean': 1.0146310329437256, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.952474370598793, 'clip_ratio/low_mean': 0.052476415410637856, 'clip_ratio/low_min': 0.052476415410637856, 'clip_ratio/high_mean': 0.17280091531574726, 'clip_ratio/high_max': 0.17280091531574726, 'clip_ratio/region_mean': 0.22527733072638512, 'reward_total_mean': 0.6797791123390198, 'reward_meter_mean': 0.6842314004898071, 'reward_meter_std': 0.4053438603878021, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 1.0, 'reward_repeat_soft_std': 0.0, 'reward_judge_quality_mean': 0.40625, 'reward_judge_quality_std': 0.0645727664232254, 'reward_total_composite_mean': 0.6797791123390198, 'reward_total_composite_std': 0.17921513319015503, 'epoch': 0.01} + 4%|▍ | 127/3300 [17:30<5:47:13, 6.57s/it]INFO 04-12 22:31:58 [block_pool.py:378] Successfully reset prefix cache + 4%|▍ | 128/3300 [17:38<6:04:11, 6.89s/it]2026-04-12 22:32:05,912 | INFO | train_grpo_train | metrics_logged mode=train step=128 + {'loss': 0.1126, 'grad_norm': 9.270278930664062, 'learning_rate': 9.615151515151517e-06, 'num_tokens': 243477.0, 'completions/mean_length': 113.875, 'completions/min_length': 93.0, 'completions/max_length': 149.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 113.875, 'completions/min_terminated_length': 93.0, 'completions/max_terminated_length': 149.0, 'rewards/meter/mean': 0.36527761816978455, 'rewards/meter/std': 0.41165366768836975, 'rewards/count_adherence/mean': 0.96875, 'rewards/count_adherence/std': 0.0883883461356163, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9943387508392334, 'rewards/repeat_soft/std': 0.004277929663658142, 'rewards/judge_quality/mean': 0.5099999904632568, 'rewards/judge_quality/std': 0.2343989461660385, 'rewards/total_composite/mean': 0.5621213316917419, 'rewards/total_composite/std': 0.23888105154037476, 'reward': 0.5621213316917419, 'reward_std': 0.23888105154037476, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.2154771238565445, 'sampling/sampling_logp_difference/max': 2.0755271911621094, 'sampling/importance_sampling_ratio/min': 0.12549026310443878, 'sampling/importance_sampling_ratio/mean': 1.047012448310852, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.288969650864601, 'clip_ratio/low_mean': 0.10721415467560291, 'clip_ratio/low_min': 0.10721415467560291, 'clip_ratio/high_mean': 0.08637929800897837, 'clip_ratio/high_max': 0.08637929800897837, 'clip_ratio/region_mean': 0.19359345268458128, 'reward_total_mean': 0.5621213316917419, 'reward_meter_mean': 0.36527761816978455, 'reward_meter_std': 0.41165366768836975, 'reward_count_adherence_mean': 0.96875, 'reward_count_adherence_std': 0.0883883461356163, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9943387508392334, 'reward_repeat_soft_std': 0.004277929663658142, 'reward_judge_quality_mean': 0.5099999904632568, 'reward_judge_quality_std': 0.2343989461660385, 'reward_total_composite_mean': 0.5621213316917419, 'reward_total_composite_std': 0.23888105154037476, 'epoch': 0.01} + 4%|▍ | 128/3300 [17:38<6:04:11, 6.89s/it]INFO 04-12 22:32:06 [block_pool.py:378] Successfully reset prefix cache + 4%|▍ | 129/3300 [17:45<6:10:55, 7.02s/it]2026-04-12 22:32:13,233 | INFO | train_grpo_train | metrics_logged mode=train step=129 + {'loss': 0.0925, 'grad_norm': 10.018613815307617, 'learning_rate': 9.612121212121212e-06, 'num_tokens': 245700.0, 'completions/mean_length': 124.875, 'completions/min_length': 76.0, 'completions/max_length': 144.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 124.875, 'completions/min_terminated_length': 76.0, 'completions/max_terminated_length': 144.0, 'rewards/meter/mean': 0.8152534365653992, 'rewards/meter/std': 0.2667849361896515, 'rewards/count_adherence/mean': 0.96875, 'rewards/count_adherence/std': 0.0883883461356163, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.995792806148529, 'rewards/repeat_soft/std': 0.003435641061514616, 'rewards/judge_quality/mean': 0.3774999976158142, 'rewards/judge_quality/std': 0.07869470119476318, 'rewards/total_composite/mean': 0.7250058650970459, 'rewards/total_composite/std': 0.13674314320087433, 'reward': 0.7250058650970459, 'reward_std': 0.13674314320087433, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.19991059601306915, 'sampling/sampling_logp_difference/max': 1.490544319152832, 'sampling/importance_sampling_ratio/min': 0.2252500206232071, 'sampling/importance_sampling_ratio/mean': 1.0232343673706055, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.2211784571409225, 'clip_ratio/low_mean': 0.049293627962470055, 'clip_ratio/low_min': 0.049293627962470055, 'clip_ratio/high_mean': 0.16842706594616175, 'clip_ratio/high_max': 0.16842706594616175, 'clip_ratio/region_mean': 0.2177206939086318, 'reward_total_mean': 0.7250058650970459, 'reward_meter_mean': 0.8152534365653992, 'reward_meter_std': 0.2667849361896515, 'reward_count_adherence_mean': 0.96875, 'reward_count_adherence_std': 0.0883883461356163, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.995792806148529, 'reward_repeat_soft_std': 0.003435641061514616, 'reward_judge_quality_mean': 0.3774999976158142, 'reward_judge_quality_std': 0.07869470119476318, 'reward_total_composite_mean': 0.7250058650970459, 'reward_total_composite_std': 0.13674314320087433, 'epoch': 0.01} + 4%|▍ | 129/3300 [17:45<6:10:55, 7.02s/it]INFO 04-12 22:32:13 [block_pool.py:378] Successfully reset prefix cache + 4%|▍ | 130/3300 [17:52<6:12:59, 7.06s/it]2026-04-12 22:32:20,389 | INFO | train_grpo_train | metrics_logged mode=train step=130 + {'loss': 0.0572, 'grad_norm': 8.2088041305542, 'learning_rate': 9.60909090909091e-06, 'num_tokens': 248142.0, 'completions/mean_length': 128.25, 'completions/min_length': 89.0, 'completions/max_length': 164.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 128.25, 'completions/min_terminated_length': 89.0, 'completions/max_terminated_length': 164.0, 'rewards/meter/mean': 0.7240664958953857, 'rewards/meter/std': 0.30356472730636597, 'rewards/count_adherence/mean': 0.96875, 'rewards/count_adherence/std': 0.0883883461356163, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9945056438446045, 'rewards/repeat_soft/std': 0.0045688156969845295, 'rewards/judge_quality/mean': 0.35624998807907104, 'rewards/judge_quality/std': 0.08798335492610931, 'rewards/total_composite/mean': 0.6774680018424988, 'rewards/total_composite/std': 0.1481560319662094, 'reward': 0.6774680018424988, 'reward_std': 0.1481560319662094, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.20742681622505188, 'sampling/sampling_logp_difference/max': 1.6332435607910156, 'sampling/importance_sampling_ratio/min': 0.1952950805425644, 'sampling/importance_sampling_ratio/mean': 1.0403975248336792, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.306053400039673, 'clip_ratio/low_mean': 0.059869431890547276, 'clip_ratio/low_min': 0.059869431890547276, 'clip_ratio/high_mean': 0.12077674269676208, 'clip_ratio/high_max': 0.12077674269676208, 'clip_ratio/region_mean': 0.18064617458730936, 'reward_total_mean': 0.6774680018424988, 'reward_meter_mean': 0.7240664958953857, 'reward_meter_std': 0.30356472730636597, 'reward_count_adherence_mean': 0.96875, 'reward_count_adherence_std': 0.0883883461356163, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9945056438446045, 'reward_repeat_soft_std': 0.0045688156969845295, 'reward_judge_quality_mean': 0.35624998807907104, 'reward_judge_quality_std': 0.08798335492610931, 'reward_total_composite_mean': 0.6774680018424988, 'reward_total_composite_std': 0.1481560319662094, 'epoch': 0.01} + 4%|▍ | 130/3300 [17:52<6:12:59, 7.06s/it]INFO 04-12 22:32:20 [block_pool.py:378] Successfully reset prefix cache + 4%|▍ | 131/3300 [17:59<6:01:18, 6.84s/it]2026-04-12 22:32:26,720 | INFO | train_grpo_train | metrics_logged mode=train step=131 + {'loss': -0.0276, 'grad_norm': 16.39191246032715, 'learning_rate': 9.606060606060607e-06, 'num_tokens': 249709.0, 'completions/mean_length': 37.875, 'completions/min_length': 28.0, 'completions/max_length': 48.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 37.875, 'completions/min_terminated_length': 28.0, 'completions/max_terminated_length': 48.0, 'rewards/meter/mean': 0.6100648641586304, 'rewards/meter/std': 0.42265385389328003, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9827827215194702, 'rewards/repeat_soft/std': 0.029818646609783173, 'rewards/judge_quality/mean': 0.5275000333786011, 'rewards/judge_quality/std': 0.2499571591615677, 'rewards/total_composite/mean': 0.6810574531555176, 'rewards/total_composite/std': 0.15360815823078156, 'reward': 0.6810574531555176, 'reward_std': 0.15360815823078156, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.2118445187807083, 'sampling/sampling_logp_difference/max': 2.0108089447021484, 'sampling/importance_sampling_ratio/min': 0.13388033211231232, 'sampling/importance_sampling_ratio/mean': 1.0341925621032715, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.3326124921441078, 'clip_ratio/low_mean': 0.11372986994683743, 'clip_ratio/low_min': 0.11372986994683743, 'clip_ratio/high_mean': 0.14517483115196228, 'clip_ratio/high_max': 0.14517483115196228, 'clip_ratio/region_mean': 0.2589047010987997, 'reward_total_mean': 0.6810574531555176, 'reward_meter_mean': 0.6100648641586304, 'reward_meter_std': 0.42265385389328003, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9827827215194702, 'reward_repeat_soft_std': 0.029818646609783173, 'reward_judge_quality_mean': 0.5275000333786011, 'reward_judge_quality_std': 0.2499571591615677, 'reward_total_composite_mean': 0.6810574531555176, 'reward_total_composite_std': 0.15360815823078156, 'epoch': 0.01} + 4%|▍ | 131/3300 [17:59<6:01:18, 6.84s/it]INFO 04-12 22:32:26 [block_pool.py:378] Successfully reset prefix cache + 4%|▍ | 132/3300 [18:06<6:07:24, 6.96s/it]2026-04-12 22:32:33,952 | INFO | train_grpo_train | metrics_logged mode=train step=132 + {'loss': 0.0561, 'grad_norm': 8.524138450622559, 'learning_rate': 9.603030303030304e-06, 'num_tokens': 252444.0, 'completions/mean_length': 138.875, 'completions/min_length': 127.0, 'completions/max_length': 160.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 138.875, 'completions/min_terminated_length': 127.0, 'completions/max_terminated_length': 160.0, 'rewards/meter/mean': 0.3716757893562317, 'rewards/meter/std': 0.3261376619338989, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9943193793296814, 'rewards/repeat_soft/std': 0.0035428176634013653, 'rewards/judge_quality/mean': 0.39375001192092896, 'rewards/judge_quality/std': 0.15638209879398346, 'rewards/total_composite/mean': 0.5348110198974609, 'rewards/total_composite/std': 0.18445618450641632, 'reward': 0.5348110198974609, 'reward_std': 0.18445616960525513, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.18236319720745087, 'sampling/sampling_logp_difference/max': 1.5875434875488281, 'sampling/importance_sampling_ratio/min': 0.20442716777324677, 'sampling/importance_sampling_ratio/mean': 1.0325572490692139, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.9220962226390839, 'clip_ratio/low_mean': 0.09894000738859177, 'clip_ratio/low_min': 0.09894000738859177, 'clip_ratio/high_mean': 0.07793621346354485, 'clip_ratio/high_max': 0.07793621346354485, 'clip_ratio/region_mean': 0.1768762208521366, 'reward_total_mean': 0.5348110198974609, 'reward_meter_mean': 0.3716757893562317, 'reward_meter_std': 0.3261376619338989, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9943193793296814, 'reward_repeat_soft_std': 0.0035428176634013653, 'reward_judge_quality_mean': 0.39375001192092896, 'reward_judge_quality_std': 0.15638209879398346, 'reward_total_composite_mean': 0.5348110198974609, 'reward_total_composite_std': 0.18445618450641632, 'epoch': 0.01} + 4%|▍ | 132/3300 [18:06<6:07:24, 6.96s/it]INFO 04-12 22:32:34 [block_pool.py:378] Successfully reset prefix cache + 4%|▍ | 133/3300 [18:13<6:14:15, 7.09s/it]2026-04-12 22:32:41,349 | INFO | train_grpo_train | metrics_logged mode=train step=133 + {'loss': 0.1566, 'grad_norm': 18.800283432006836, 'learning_rate': 9.600000000000001e-06, 'num_tokens': 254891.0, 'completions/mean_length': 106.875, 'completions/min_length': 84.0, 'completions/max_length': 143.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 106.875, 'completions/min_terminated_length': 84.0, 'completions/max_terminated_length': 143.0, 'rewards/meter/mean': 0.9262809753417969, 'rewards/meter/std': 0.07574860751628876, 'rewards/count_adherence/mean': 0.875, 'rewards/count_adherence/std': 0.24800792336463928, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9938892126083374, 'rewards/repeat_soft/std': 0.008599473163485527, 'rewards/judge_quality/mean': 0.42374998331069946, 'rewards/judge_quality/std': 0.010606602765619755, 'rewards/total_composite/mean': 0.7745903730392456, 'rewards/total_composite/std': 0.05820745229721069, 'reward': 0.7745903730392456, 'reward_std': 0.05820745974779129, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.18495529890060425, 'sampling/sampling_logp_difference/max': 1.525949478149414, 'sampling/importance_sampling_ratio/min': 0.21741452813148499, 'sampling/importance_sampling_ratio/mean': 1.0175228118896484, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.949052482843399, 'clip_ratio/low_mean': 0.0336101409047842, 'clip_ratio/low_min': 0.0336101409047842, 'clip_ratio/high_mean': 0.145680645480752, 'clip_ratio/high_max': 0.145680645480752, 'clip_ratio/region_mean': 0.1792907863855362, 'reward_total_mean': 0.7745903730392456, 'reward_meter_mean': 0.9262809753417969, 'reward_meter_std': 0.07574860751628876, 'reward_count_adherence_mean': 0.875, 'reward_count_adherence_std': 0.24800792336463928, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9938892126083374, 'reward_repeat_soft_std': 0.008599473163485527, 'reward_judge_quality_mean': 0.42374998331069946, 'reward_judge_quality_std': 0.010606602765619755, 'reward_total_composite_mean': 0.7745903730392456, 'reward_total_composite_std': 0.05820745229721069, 'epoch': 0.01} + 4%|▍ | 133/3300 [18:13<6:14:15, 7.09s/it]INFO 04-12 22:32:41 [block_pool.py:378] Successfully reset prefix cache + 4%|▍ | 134/3300 [18:21<6:23:32, 7.27s/it]2026-04-12 22:32:49,035 | INFO | train_grpo_train | metrics_logged mode=train step=134 + {'loss': 0.0607, 'grad_norm': 7.037054538726807, 'learning_rate': 9.596969696969699e-06, 'num_tokens': 257734.0, 'completions/mean_length': 162.375, 'completions/min_length': 123.0, 'completions/max_length': 186.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 162.375, 'completions/min_terminated_length': 123.0, 'completions/max_terminated_length': 186.0, 'rewards/meter/mean': 0.9117918014526367, 'rewards/meter/std': 0.22455473244190216, 'rewards/count_adherence/mean': 0.9750000238418579, 'rewards/count_adherence/std': 0.0707106739282608, 'rewards/hard_gate/mean': 0.75, 'rewards/hard_gate/std': 0.4629100561141968, 'rewards/repeat_soft/mean': 0.9934555292129517, 'rewards/repeat_soft/std': 0.005569420754909515, 'rewards/judge_quality/mean': 0.41499999165534973, 'rewards/judge_quality/std': 0.14520922303199768, 'rewards/total_composite/mean': 0.587673008441925, 'rewards/total_composite/std': 0.3813877999782562, 'reward': 0.587673008441925, 'reward_std': 0.38138777017593384, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.19455531239509583, 'sampling/sampling_logp_difference/max': 1.505533218383789, 'sampling/importance_sampling_ratio/min': 0.22189894318580627, 'sampling/importance_sampling_ratio/mean': 1.0390830039978027, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.4518264532089233, 'clip_ratio/low_mean': 0.0657502580434084, 'clip_ratio/low_min': 0.0657502580434084, 'clip_ratio/high_mean': 0.12715561501681805, 'clip_ratio/high_max': 0.12715561501681805, 'clip_ratio/region_mean': 0.19290587306022644, 'reward_total_mean': 0.587673008441925, 'reward_meter_mean': 0.9117918014526367, 'reward_meter_std': 0.22455473244190216, 'reward_count_adherence_mean': 0.9750000238418579, 'reward_count_adherence_std': 0.0707106739282608, 'reward_hard_gate_mean': 0.75, 'reward_hard_gate_std': 0.4629100561141968, 'reward_repeat_soft_mean': 0.9934555292129517, 'reward_repeat_soft_std': 0.005569420754909515, 'reward_judge_quality_mean': 0.41499999165534973, 'reward_judge_quality_std': 0.14520922303199768, 'reward_total_composite_mean': 0.587673008441925, 'reward_total_composite_std': 0.3813877999782562, 'epoch': 0.01} + 4%|▍ | 134/3300 [18:21<6:23:32, 7.27s/it]INFO 04-12 22:32:49 [block_pool.py:378] Successfully reset prefix cache + 4%|▍ | 135/3300 [18:29<6:28:03, 7.36s/it]2026-04-12 22:32:56,600 | INFO | train_grpo_train | metrics_logged mode=train step=135 + {'loss': 0.0783, 'grad_norm': 7.698927879333496, 'learning_rate': 9.593939393939394e-06, 'num_tokens': 260125.0, 'completions/mean_length': 132.875, 'completions/min_length': 86.0, 'completions/max_length': 170.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 132.875, 'completions/min_terminated_length': 86.0, 'completions/max_terminated_length': 170.0, 'rewards/meter/mean': 0.512225329875946, 'rewards/meter/std': 0.2971871495246887, 'rewards/count_adherence/mean': 0.96875, 'rewards/count_adherence/std': 0.0883883461356163, 'rewards/hard_gate/mean': 0.875, 'rewards/hard_gate/std': 0.3535533845424652, 'rewards/repeat_soft/mean': 0.9965784549713135, 'rewards/repeat_soft/std': 0.003520975122228265, 'rewards/judge_quality/mean': 0.42750000953674316, 'rewards/judge_quality/std': 0.22403764724731445, 'rewards/total_composite/mean': 0.5387158393859863, 'rewards/total_composite/std': 0.27052682638168335, 'reward': 0.5387158393859863, 'reward_std': 0.27052679657936096, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.20126554369926453, 'sampling/sampling_logp_difference/max': 1.7666950225830078, 'sampling/importance_sampling_ratio/min': 0.17089687287807465, 'sampling/importance_sampling_ratio/mean': 1.0329970121383667, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.3159025609493256, 'clip_ratio/low_mean': 0.0917290709912777, 'clip_ratio/low_min': 0.0917290709912777, 'clip_ratio/high_mean': 0.11301391199231148, 'clip_ratio/high_max': 0.11301391199231148, 'clip_ratio/region_mean': 0.20474298298358917, 'reward_total_mean': 0.5387158393859863, 'reward_meter_mean': 0.512225329875946, 'reward_meter_std': 0.2971871495246887, 'reward_count_adherence_mean': 0.96875, 'reward_count_adherence_std': 0.0883883461356163, 'reward_hard_gate_mean': 0.875, 'reward_hard_gate_std': 0.3535533845424652, 'reward_repeat_soft_mean': 0.9965784549713135, 'reward_repeat_soft_std': 0.003520975122228265, 'reward_judge_quality_mean': 0.42750000953674316, 'reward_judge_quality_std': 0.22403764724731445, 'reward_total_composite_mean': 0.5387158393859863, 'reward_total_composite_std': 0.27052682638168335, 'epoch': 0.01} + 4%|▍ | 135/3300 [18:29<6:28:03, 7.36s/it]INFO 04-12 22:32:56 [block_pool.py:378] Successfully reset prefix cache + 4%|▍ | 136/3300 [18:36<6:23:55, 7.28s/it]2026-04-12 22:33:03,699 | INFO | train_grpo_train | metrics_logged mode=train step=136 + {'loss': 0.0376, 'grad_norm': 9.772281646728516, 'learning_rate': 9.590909090909091e-06, 'num_tokens': 262290.0, 'completions/mean_length': 90.625, 'completions/min_length': 76.0, 'completions/max_length': 119.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 90.625, 'completions/min_terminated_length': 76.0, 'completions/max_terminated_length': 119.0, 'rewards/meter/mean': 0.5402966737747192, 'rewards/meter/std': 0.4211829900741577, 'rewards/count_adherence/mean': 0.9583333730697632, 'rewards/count_adherence/std': 0.117851123213768, 'rewards/hard_gate/mean': 0.875, 'rewards/hard_gate/std': 0.3535533845424652, 'rewards/repeat_soft/mean': 0.984871506690979, 'rewards/repeat_soft/std': 0.013639903627336025, 'rewards/judge_quality/mean': 0.5824999809265137, 'rewards/judge_quality/std': 0.23260943591594696, 'rewards/total_composite/mean': 0.6112963557243347, 'rewards/total_composite/std': 0.3185757100582123, 'reward': 0.6112963557243347, 'reward_std': 0.3185757100582123, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.20179298520088196, 'sampling/sampling_logp_difference/max': 1.3335742950439453, 'sampling/importance_sampling_ratio/min': 0.26353365182876587, 'sampling/importance_sampling_ratio/mean': 0.9979775547981262, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.586782455444336, 'clip_ratio/low_mean': 0.05890737473964691, 'clip_ratio/low_min': 0.05890737473964691, 'clip_ratio/high_mean': 0.1278254184871912, 'clip_ratio/high_max': 0.1278254184871912, 'clip_ratio/region_mean': 0.1867327932268381, 'reward_total_mean': 0.6112963557243347, 'reward_meter_mean': 0.5402966737747192, 'reward_meter_std': 0.4211829900741577, 'reward_count_adherence_mean': 0.9583333730697632, 'reward_count_adherence_std': 0.117851123213768, 'reward_hard_gate_mean': 0.875, 'reward_hard_gate_std': 0.3535533845424652, 'reward_repeat_soft_mean': 0.984871506690979, 'reward_repeat_soft_std': 0.013639903627336025, 'reward_judge_quality_mean': 0.5824999809265137, 'reward_judge_quality_std': 0.23260943591594696, 'reward_total_composite_mean': 0.6112963557243347, 'reward_total_composite_std': 0.3185757100582123, 'epoch': 0.01} + 4%|▍ | 136/3300 [18:36<6:23:55, 7.28s/it]INFO 04-12 22:33:03 [block_pool.py:378] Successfully reset prefix cache + 4%|▍ | 137/3300 [18:42<6:05:29, 6.93s/it]2026-04-12 22:33:09,821 | INFO | train_grpo_train | metrics_logged mode=train step=137 + {'loss': -0.0179, 'grad_norm': 15.270221710205078, 'learning_rate': 9.587878787878789e-06, 'num_tokens': 263927.0, 'completions/mean_length': 50.625, 'completions/min_length': 38.0, 'completions/max_length': 59.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 50.625, 'completions/min_terminated_length': 38.0, 'completions/max_terminated_length': 59.0, 'rewards/meter/mean': 0.12983430922031403, 'rewards/meter/std': 0.1328689157962799, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 0.75, 'rewards/hard_gate/std': 0.4629100561141968, 'rewards/repeat_soft/mean': 0.9958950877189636, 'rewards/repeat_soft/std': 0.005419950000941753, 'rewards/judge_quality/mean': 0.5275000333786011, 'rewards/judge_quality/std': 0.2499571591615677, 'rewards/total_composite/mean': 0.37045469880104065, 'rewards/total_composite/std': 0.23874065279960632, 'reward': 0.37045469880104065, 'reward_std': 0.23874063789844513, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.20379044115543365, 'sampling/sampling_logp_difference/max': 1.362325668334961, 'sampling/importance_sampling_ratio/min': 0.2560645639896393, 'sampling/importance_sampling_ratio/mean': 1.0346601009368896, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.0476372092962265, 'clip_ratio/low_mean': 0.05046296305954456, 'clip_ratio/low_min': 0.05046296305954456, 'clip_ratio/high_mean': 0.1526920609176159, 'clip_ratio/high_max': 0.1526920609176159, 'clip_ratio/region_mean': 0.20315502397716045, 'reward_total_mean': 0.37045469880104065, 'reward_meter_mean': 0.12983430922031403, 'reward_meter_std': 0.1328689157962799, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 0.75, 'reward_hard_gate_std': 0.4629100561141968, 'reward_repeat_soft_mean': 0.9958950877189636, 'reward_repeat_soft_std': 0.005419950000941753, 'reward_judge_quality_mean': 0.5275000333786011, 'reward_judge_quality_std': 0.2499571591615677, 'reward_total_composite_mean': 0.37045469880104065, 'reward_total_composite_std': 0.23874065279960632, 'epoch': 0.01} + 4%|▍ | 137/3300 [18:42<6:05:29, 6.93s/it]INFO 04-12 22:33:10 [block_pool.py:378] Successfully reset prefix cache + 4%|▍ | 138/3300 [18:48<5:51:30, 6.67s/it]2026-04-12 22:33:15,876 | INFO | train_grpo_train | metrics_logged mode=train step=138 + {'loss': -0.039, 'grad_norm': 17.487319946289062, 'learning_rate': 9.584848484848486e-06, 'num_tokens': 265676.0, 'completions/mean_length': 49.625, 'completions/min_length': 43.0, 'completions/max_length': 60.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 49.625, 'completions/min_terminated_length': 43.0, 'completions/max_terminated_length': 60.0, 'rewards/meter/mean': 0.378584086894989, 'rewards/meter/std': 0.35492274165153503, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 0.875, 'rewards/hard_gate/std': 0.3535533845424652, 'rewards/repeat_soft/mean': 0.9902465343475342, 'rewards/repeat_soft/std': 0.011248313821852207, 'rewards/judge_quality/mean': 0.6075000166893005, 'rewards/judge_quality/std': 0.25877460837364197, 'rewards/total_composite/mean': 0.5539078712463379, 'rewards/total_composite/std': 0.26821252703666687, 'reward': 0.5539078712463379, 'reward_std': 0.26821252703666687, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.2223401814699173, 'sampling/sampling_logp_difference/max': 2.262995719909668, 'sampling/importance_sampling_ratio/min': 0.10403835028409958, 'sampling/importance_sampling_ratio/mean': 1.0410206317901611, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.078471839427948, 'clip_ratio/low_mean': 0.12089350260794163, 'clip_ratio/low_min': 0.12089350260794163, 'clip_ratio/high_mean': 0.1181613514199853, 'clip_ratio/high_max': 0.1181613514199853, 'clip_ratio/region_mean': 0.23905485402792692, 'reward_total_mean': 0.5539078712463379, 'reward_meter_mean': 0.378584086894989, 'reward_meter_std': 0.35492274165153503, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 0.875, 'reward_hard_gate_std': 0.3535533845424652, 'reward_repeat_soft_mean': 0.9902465343475342, 'reward_repeat_soft_std': 0.011248313821852207, 'reward_judge_quality_mean': 0.6075000166893005, 'reward_judge_quality_std': 0.25877460837364197, 'reward_total_composite_mean': 0.5539078712463379, 'reward_total_composite_std': 0.26821252703666687, 'epoch': 0.01} + 4%|▍ | 138/3300 [18:48<5:51:30, 6.67s/it]INFO 04-12 22:33:16 [block_pool.py:378] Successfully reset prefix cache + 4%|▍ | 139/3300 [18:55<5:51:23, 6.67s/it]2026-04-12 22:33:22,548 | INFO | train_grpo_train | metrics_logged mode=train step=139 + {'loss': 0.0251, 'grad_norm': 10.91278076171875, 'learning_rate': 9.581818181818181e-06, 'num_tokens': 267487.0, 'completions/mean_length': 75.375, 'completions/min_length': 63.0, 'completions/max_length': 98.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 75.375, 'completions/min_terminated_length': 63.0, 'completions/max_terminated_length': 98.0, 'rewards/meter/mean': 0.7106475830078125, 'rewards/meter/std': 0.3840155303478241, 'rewards/count_adherence/mean': 0.9375, 'rewards/count_adherence/std': 0.1767766922712326, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9995355010032654, 'rewards/repeat_soft/std': 0.000715656962711364, 'rewards/judge_quality/mean': 0.4674999713897705, 'rewards/judge_quality/std': 0.179344043135643, 'rewards/total_composite/mean': 0.7006199955940247, 'rewards/total_composite/std': 0.19790136814117432, 'reward': 0.7006199955940247, 'reward_std': 0.1979013830423355, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.1969791203737259, 'sampling/sampling_logp_difference/max': 1.579707145690918, 'sampling/importance_sampling_ratio/min': 0.20603543519973755, 'sampling/importance_sampling_ratio/mean': 1.0220390558242798, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.9457141309976578, 'clip_ratio/low_mean': 0.0600130520761013, 'clip_ratio/low_min': 0.0600130520761013, 'clip_ratio/high_mean': 0.1256153592839837, 'clip_ratio/high_max': 0.1256153592839837, 'clip_ratio/region_mean': 0.185628411360085, 'reward_total_mean': 0.7006199955940247, 'reward_meter_mean': 0.7106475830078125, 'reward_meter_std': 0.3840155303478241, 'reward_count_adherence_mean': 0.9375, 'reward_count_adherence_std': 0.1767766922712326, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9995355010032654, 'reward_repeat_soft_std': 0.000715656962711364, 'reward_judge_quality_mean': 0.4674999713897705, 'reward_judge_quality_std': 0.179344043135643, 'reward_total_composite_mean': 0.7006199955940247, 'reward_total_composite_std': 0.19790136814117432, 'epoch': 0.01} + 4%|▍ | 139/3300 [18:55<5:51:23, 6.67s/it]INFO 04-12 22:33:22 [block_pool.py:378] Successfully reset prefix cache + 4%|▍ | 140/3300 [19:00<5:39:44, 6.45s/it]2026-04-12 22:33:28,488 | INFO | train_grpo_train | metrics_logged mode=train step=140 + {'loss': -0.0163, 'grad_norm': 24.94887924194336, 'learning_rate': 9.57878787878788e-06, 'num_tokens': 269104.0, 'completions/mean_length': 44.125, 'completions/min_length': 39.0, 'completions/max_length': 47.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 44.125, 'completions/min_terminated_length': 39.0, 'completions/max_terminated_length': 47.0, 'rewards/meter/mean': 0.8744622468948364, 'rewards/meter/std': 0.31945866346359253, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9869142174720764, 'rewards/repeat_soft/std': 0.0278518907725811, 'rewards/judge_quality/mean': 0.5237500071525574, 'rewards/judge_quality/std': 0.25150617957115173, 'rewards/total_composite/mean': 0.7993244528770447, 'rewards/total_composite/std': 0.1901601254940033, 'reward': 0.7993244528770447, 'reward_std': 0.19016008079051971, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.14177052676677704, 'sampling/sampling_logp_difference/max': 2.6750831604003906, 'sampling/importance_sampling_ratio/min': 0.06890109926462173, 'sampling/importance_sampling_ratio/mean': 0.9756723642349243, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 0.7793125845491886, 'clip_ratio/low_mean': 0.012820512987673283, 'clip_ratio/low_min': 0.012820512987673283, 'clip_ratio/high_mean': 0.11867795884609222, 'clip_ratio/high_max': 0.11867795884609222, 'clip_ratio/region_mean': 0.1314984718337655, 'reward_total_mean': 0.7993244528770447, 'reward_meter_mean': 0.8744622468948364, 'reward_meter_std': 0.31945866346359253, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9869142174720764, 'reward_repeat_soft_std': 0.0278518907725811, 'reward_judge_quality_mean': 0.5237500071525574, 'reward_judge_quality_std': 0.25150617957115173, 'reward_total_composite_mean': 0.7993244528770447, 'reward_total_composite_std': 0.1901601254940033, 'epoch': 0.01} + 4%|▍ | 140/3300 [19:00<5:39:44, 6.45s/it]INFO 04-12 22:33:28 [block_pool.py:378] Successfully reset prefix cache + 4%|▍ | 141/3300 [19:09<6:10:39, 7.04s/it]2026-04-12 22:33:36,902 | INFO | train_grpo_train | metrics_logged mode=train step=141 + {'loss': 0.0086, 'grad_norm': 7.010746955871582, 'learning_rate': 9.575757575757576e-06, 'num_tokens': 272341.0, 'completions/mean_length': 198.625, 'completions/min_length': 128.0, 'completions/max_length': 243.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 198.625, 'completions/min_terminated_length': 128.0, 'completions/max_terminated_length': 243.0, 'rewards/meter/mean': 0.45272260904312134, 'rewards/meter/std': 0.27589282393455505, 'rewards/count_adherence/mean': 0.9166666269302368, 'rewards/count_adherence/std': 0.1259881556034088, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9925938844680786, 'rewards/repeat_soft/std': 0.005698754917830229, 'rewards/judge_quality/mean': 0.35624998807907104, 'rewards/judge_quality/std': 0.08798335492610931, 'rewards/total_composite/mean': 0.5473595857620239, 'rewards/total_composite/std': 0.14514927566051483, 'reward': 0.5473595857620239, 'reward_std': 0.14514927566051483, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.18950483202934265, 'sampling/sampling_logp_difference/max': 2.614901542663574, 'sampling/importance_sampling_ratio/min': 0.07317499071359634, 'sampling/importance_sampling_ratio/mean': 1.0239454507827759, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.8596653640270233, 'clip_ratio/low_mean': 0.11194692924618721, 'clip_ratio/low_min': 0.11194692924618721, 'clip_ratio/high_mean': 0.06177238188683987, 'clip_ratio/high_max': 0.06177238188683987, 'clip_ratio/region_mean': 0.17371931113302708, 'reward_total_mean': 0.5473595857620239, 'reward_meter_mean': 0.45272260904312134, 'reward_meter_std': 0.27589282393455505, 'reward_count_adherence_mean': 0.9166666269302368, 'reward_count_adherence_std': 0.1259881556034088, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9925938844680786, 'reward_repeat_soft_std': 0.005698754917830229, 'reward_judge_quality_mean': 0.35624998807907104, 'reward_judge_quality_std': 0.08798335492610931, 'reward_total_composite_mean': 0.5473595857620239, 'reward_total_composite_std': 0.14514927566051483, 'epoch': 0.01} + 4%|▍ | 141/3300 [19:09<6:10:39, 7.04s/it]INFO 04-12 22:33:37 [block_pool.py:378] Successfully reset prefix cache + 4%|▍ | 142/3300 [19:15<6:02:42, 6.89s/it]2026-04-12 22:33:43,445 | INFO | train_grpo_train | metrics_logged mode=train step=142 + {'loss': 0.0003, 'grad_norm': 12.71095085144043, 'learning_rate': 9.572727272727273e-06, 'num_tokens': 274053.0, 'completions/mean_length': 55.0, 'completions/min_length': 45.0, 'completions/max_length': 63.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 55.0, 'completions/min_terminated_length': 45.0, 'completions/max_terminated_length': 63.0, 'rewards/meter/mean': 0.7612498998641968, 'rewards/meter/std': 0.29422006011009216, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 1.0, 'rewards/repeat_soft/std': 0.0, 'rewards/judge_quality/mean': 0.5149999856948853, 'rewards/judge_quality/std': 0.16017849743366241, 'rewards/total_composite/mean': 0.7470624446868896, 'rewards/total_composite/std': 0.13542601466178894, 'reward': 0.7470624446868896, 'reward_std': 0.13542599976062775, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.21361897885799408, 'sampling/sampling_logp_difference/max': 1.2875089645385742, 'sampling/importance_sampling_ratio/min': 0.2759573459625244, 'sampling/importance_sampling_ratio/mean': 1.0175107717514038, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.415143460035324, 'clip_ratio/low_mean': 0.06529347132891417, 'clip_ratio/low_min': 0.06529347132891417, 'clip_ratio/high_mean': 0.11970782186836004, 'clip_ratio/high_max': 0.11970782186836004, 'clip_ratio/region_mean': 0.1850012931972742, 'reward_total_mean': 0.7470624446868896, 'reward_meter_mean': 0.7612498998641968, 'reward_meter_std': 0.29422006011009216, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 1.0, 'reward_repeat_soft_std': 0.0, 'reward_judge_quality_mean': 0.5149999856948853, 'reward_judge_quality_std': 0.16017849743366241, 'reward_total_composite_mean': 0.7470624446868896, 'reward_total_composite_std': 0.13542601466178894, 'epoch': 0.01} + 4%|▍ | 142/3300 [19:15<6:02:42, 6.89s/it]INFO 04-12 22:33:43 [block_pool.py:378] Successfully reset prefix cache + 4%|▍ | 143/3300 [19:23<6:16:52, 7.16s/it]2026-04-12 22:33:51,242 | INFO | train_grpo_train | metrics_logged mode=train step=143 + {'loss': 0.0499, 'grad_norm': 7.3252692222595215, 'learning_rate': 9.56969696969697e-06, 'num_tokens': 277000.0, 'completions/mean_length': 181.375, 'completions/min_length': 156.0, 'completions/max_length': 205.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 181.375, 'completions/min_terminated_length': 156.0, 'completions/max_terminated_length': 205.0, 'rewards/meter/mean': 0.6341123580932617, 'rewards/meter/std': 0.3794861435890198, 'rewards/count_adherence/mean': 0.9791666269302368, 'rewards/count_adherence/std': 0.0589255727827549, 'rewards/hard_gate/mean': 0.75, 'rewards/hard_gate/std': 0.4629100561141968, 'rewards/repeat_soft/mean': 0.9955697059631348, 'rewards/repeat_soft/std': 0.0051034800708293915, 'rewards/judge_quality/mean': 0.3137499690055847, 'rewards/judge_quality/std': 0.08798335492610931, 'rewards/total_composite/mean': 0.4888369143009186, 'rewards/total_composite/std': 0.32723328471183777, 'reward': 0.4888369143009186, 'reward_std': 0.3272332549095154, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.21756555140018463, 'sampling/sampling_logp_difference/max': 1.8546981811523438, 'sampling/importance_sampling_ratio/min': 0.15650016069412231, 'sampling/importance_sampling_ratio/mean': 1.031713604927063, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.53104504942894, 'clip_ratio/low_mean': 0.060815850272774696, 'clip_ratio/low_min': 0.060815850272774696, 'clip_ratio/high_mean': 0.13803689181804657, 'clip_ratio/high_max': 0.13803689181804657, 'clip_ratio/region_mean': 0.19885274209082127, 'reward_total_mean': 0.4888369143009186, 'reward_meter_mean': 0.6341123580932617, 'reward_meter_std': 0.3794861435890198, 'reward_count_adherence_mean': 0.9791666269302368, 'reward_count_adherence_std': 0.0589255727827549, 'reward_hard_gate_mean': 0.75, 'reward_hard_gate_std': 0.4629100561141968, 'reward_repeat_soft_mean': 0.9955697059631348, 'reward_repeat_soft_std': 0.0051034800708293915, 'reward_judge_quality_mean': 0.3137499690055847, 'reward_judge_quality_std': 0.08798335492610931, 'reward_total_composite_mean': 0.4888369143009186, 'reward_total_composite_std': 0.32723328471183777, 'epoch': 0.01} + 4%|▍ | 143/3300 [19:23<6:16:52, 7.16s/it]INFO 04-12 22:33:51 [block_pool.py:378] Successfully reset prefix cache + 4%|▍ | 144/3300 [19:30<6:16:45, 7.16s/it]2026-04-12 22:33:58,406 | INFO | train_grpo_train | metrics_logged mode=train step=144 + {'loss': 0.0816, 'grad_norm': 12.265125274658203, 'learning_rate': 9.566666666666668e-06, 'num_tokens': 278838.0, 'completions/mean_length': 57.75, 'completions/min_length': 53.0, 'completions/max_length': 63.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 57.75, 'completions/min_terminated_length': 53.0, 'completions/max_terminated_length': 63.0, 'rewards/meter/mean': 0.5257716774940491, 'rewards/meter/std': 0.47132599353790283, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9956430792808533, 'rewards/repeat_soft/std': 0.0068510351702570915, 'rewards/judge_quality/mean': 0.46875, 'rewards/judge_quality/std': 0.19334925711154938, 'rewards/total_composite/mean': 0.6267865896224976, 'rewards/total_composite/std': 0.20718355476856232, 'reward': 0.6267865896224976, 'reward_std': 0.20718353986740112, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.17005670070648193, 'sampling/sampling_logp_difference/max': 1.1533842086791992, 'sampling/importance_sampling_ratio/min': 0.3155670166015625, 'sampling/importance_sampling_ratio/mean': 1.0278751850128174, 'sampling/importance_sampling_ratio/max': 1.8527933359146118, 'entropy': 1.7537786066532135, 'clip_ratio/low_mean': 0.09149759635329247, 'clip_ratio/low_min': 0.09149759635329247, 'clip_ratio/high_mean': 0.09468746930360794, 'clip_ratio/high_max': 0.09468746930360794, 'clip_ratio/region_mean': 0.1861850656569004, 'reward_total_mean': 0.6267865896224976, 'reward_meter_mean': 0.5257716774940491, 'reward_meter_std': 0.47132599353790283, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9956430792808533, 'reward_repeat_soft_std': 0.0068510351702570915, 'reward_judge_quality_mean': 0.46875, 'reward_judge_quality_std': 0.19334925711154938, 'reward_total_composite_mean': 0.6267865896224976, 'reward_total_composite_std': 0.20718355476856232, 'epoch': 0.01} + 4%|▍ | 144/3300 [19:30<6:16:45, 7.16s/it]INFO 04-12 22:33:58 [block_pool.py:378] Successfully reset prefix cache + 4%|▍ | 145/3300 [19:37<6:03:21, 6.91s/it]2026-04-12 22:34:04,726 | INFO | train_grpo_train | metrics_logged mode=train step=145 + {'loss': 0.0394, 'grad_norm': 15.574225425720215, 'learning_rate': 9.563636363636365e-06, 'num_tokens': 280473.0, 'completions/mean_length': 49.375, 'completions/min_length': 32.0, 'completions/max_length': 66.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 49.375, 'completions/min_terminated_length': 32.0, 'completions/max_terminated_length': 66.0, 'rewards/meter/mean': 0.4346250593662262, 'rewards/meter/std': 0.4319024384021759, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9841232299804688, 'rewards/repeat_soft/std': 0.01157012116163969, 'rewards/judge_quality/mean': 0.6737500429153442, 'rewards/judge_quality/std': 0.263435423374176, 'rewards/total_composite/mean': 0.6461185812950134, 'rewards/total_composite/std': 0.2213435173034668, 'reward': 0.6461185812950134, 'reward_std': 0.2213435173034668, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.1840735524892807, 'sampling/sampling_logp_difference/max': 1.6395959854125977, 'sampling/importance_sampling_ratio/min': 0.19405843317508698, 'sampling/importance_sampling_ratio/mean': 1.062300443649292, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.101605921983719, 'clip_ratio/low_mean': 0.13492169696837664, 'clip_ratio/low_min': 0.13492169696837664, 'clip_ratio/high_mean': 0.06772094406187534, 'clip_ratio/high_max': 0.06772094406187534, 'clip_ratio/region_mean': 0.20264264103025198, 'reward_total_mean': 0.6461185812950134, 'reward_meter_mean': 0.4346250593662262, 'reward_meter_std': 0.4319024384021759, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9841232299804688, 'reward_repeat_soft_std': 0.01157012116163969, 'reward_judge_quality_mean': 0.6737500429153442, 'reward_judge_quality_std': 0.263435423374176, 'reward_total_composite_mean': 0.6461185812950134, 'reward_total_composite_std': 0.2213435173034668, 'epoch': 0.01} + 4%|▍ | 145/3300 [19:37<6:03:21, 6.91s/it]INFO 04-12 22:34:04 [block_pool.py:378] Successfully reset prefix cache + 4%|▍ | 146/3300 [19:43<5:51:56, 6.70s/it]2026-04-12 22:34:10,922 | INFO | train_grpo_train | metrics_logged mode=train step=146 + {'loss': 0.0065, 'grad_norm': 14.454992294311523, 'learning_rate': 9.56060606060606e-06, 'num_tokens': 282280.0, 'completions/mean_length': 56.875, 'completions/min_length': 41.0, 'completions/max_length': 64.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 56.875, 'completions/min_terminated_length': 41.0, 'completions/max_terminated_length': 64.0, 'rewards/meter/mean': 0.38828152418136597, 'rewards/meter/std': 0.3594237267971039, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9920034408569336, 'rewards/repeat_soft/std': 0.016578910872340202, 'rewards/judge_quality/mean': 0.41999998688697815, 'rewards/judge_quality/std': 0.0, 'rewards/total_composite/mean': 0.5499269962310791, 'rewards/total_composite/std': 0.16232407093048096, 'reward': 0.5499269962310791, 'reward_std': 0.16232407093048096, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.20530828833580017, 'sampling/sampling_logp_difference/max': 1.7603940963745117, 'sampling/importance_sampling_ratio/min': 0.17197707295417786, 'sampling/importance_sampling_ratio/mean': 1.022547721862793, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.83353291451931, 'clip_ratio/low_mean': 0.0978974774479866, 'clip_ratio/low_min': 0.0978974774479866, 'clip_ratio/high_mean': 0.10358610190451145, 'clip_ratio/high_max': 0.10358610190451145, 'clip_ratio/region_mean': 0.20148357935249805, 'reward_total_mean': 0.5499269962310791, 'reward_meter_mean': 0.38828152418136597, 'reward_meter_std': 0.3594237267971039, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9920034408569336, 'reward_repeat_soft_std': 0.016578910872340202, 'reward_judge_quality_mean': 0.41999998688697815, 'reward_judge_quality_std': 0.0, 'reward_total_composite_mean': 0.5499269962310791, 'reward_total_composite_std': 0.16232407093048096, 'epoch': 0.01} + 4%|▍ | 146/3300 [19:43<5:51:56, 6.70s/it]INFO 04-12 22:34:11 [block_pool.py:378] Successfully reset prefix cache + 4%|▍ | 147/3300 [19:49<5:43:43, 6.54s/it]2026-04-12 22:34:17,100 | INFO | train_grpo_train | metrics_logged mode=train step=147 + {'loss': 0.317, 'grad_norm': 22.578855514526367, 'learning_rate': 9.55757575757576e-06, 'num_tokens': 283814.0, 'completions/mean_length': 34.75, 'completions/min_length': 24.0, 'completions/max_length': 65.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 34.75, 'completions/min_terminated_length': 24.0, 'completions/max_terminated_length': 65.0, 'rewards/meter/mean': 0.639831006526947, 'rewards/meter/std': 0.40390339493751526, 'rewards/count_adherence/mean': 1.0, 'rewards/count_adherence/std': 0.0, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.986702561378479, 'rewards/repeat_soft/std': 0.016451282426714897, 'rewards/judge_quality/mean': 0.5824999809265137, 'rewards/judge_quality/std': 0.23260943591594696, 'rewards/total_composite/mean': 0.7113442420959473, 'rewards/total_composite/std': 0.17334672808647156, 'reward': 0.7113442420959473, 'reward_std': 0.17334672808647156, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.24534133076667786, 'sampling/sampling_logp_difference/max': 1.3638458251953125, 'sampling/importance_sampling_ratio/min': 0.2556755840778351, 'sampling/importance_sampling_ratio/mean': 1.0255770683288574, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.8735971450805664, 'clip_ratio/low_mean': 0.047843825072050095, 'clip_ratio/low_min': 0.047843825072050095, 'clip_ratio/high_mean': 0.16567365266382694, 'clip_ratio/high_max': 0.16567365266382694, 'clip_ratio/region_mean': 0.21351747773587704, 'reward_total_mean': 0.7113442420959473, 'reward_meter_mean': 0.639831006526947, 'reward_meter_std': 0.40390339493751526, 'reward_count_adherence_mean': 1.0, 'reward_count_adherence_std': 0.0, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.986702561378479, 'reward_repeat_soft_std': 0.016451282426714897, 'reward_judge_quality_mean': 0.5824999809265137, 'reward_judge_quality_std': 0.23260943591594696, 'reward_total_composite_mean': 0.7113442420959473, 'reward_total_composite_std': 0.17334672808647156, 'epoch': 0.01} + 4%|▍ | 147/3300 [19:49<5:43:43, 6.54s/it]INFO 04-12 22:34:17 [block_pool.py:378] Successfully reset prefix cache + 4%|▍ | 148/3300 [19:57<6:06:15, 6.97s/it]2026-04-12 22:34:25,078 | INFO | train_grpo_train | metrics_logged mode=train step=148 + {'loss': 0.0468, 'grad_norm': 8.733606338500977, 'learning_rate': 9.554545454545455e-06, 'num_tokens': 286450.0, 'completions/mean_length': 140.5, 'completions/min_length': 83.0, 'completions/max_length': 186.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 140.5, 'completions/min_terminated_length': 83.0, 'completions/max_terminated_length': 186.0, 'rewards/meter/mean': 0.8556537628173828, 'rewards/meter/std': 0.34665367007255554, 'rewards/count_adherence/mean': 0.949999988079071, 'rewards/count_adherence/std': 0.09258200973272324, 'rewards/hard_gate/mean': 0.875, 'rewards/hard_gate/std': 0.3535533845424652, 'rewards/repeat_soft/mean': 0.9829007983207703, 'rewards/repeat_soft/std': 0.016816018149256706, 'rewards/judge_quality/mean': 0.39374998211860657, 'rewards/judge_quality/std': 0.15638209879398346, 'rewards/total_composite/mean': 0.6418274641036987, 'rewards/total_composite/std': 0.3184812068939209, 'reward': 0.6418274641036987, 'reward_std': 0.3184812068939209, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.1881764531135559, 'sampling/sampling_logp_difference/max': 1.4606990814208984, 'sampling/importance_sampling_ratio/min': 0.23207399249076843, 'sampling/importance_sampling_ratio/mean': 1.030596137046814, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.944037601351738, 'clip_ratio/low_mean': 0.04185550473630428, 'clip_ratio/low_min': 0.04185550473630428, 'clip_ratio/high_mean': 0.11391049064695835, 'clip_ratio/high_max': 0.11391049064695835, 'clip_ratio/region_mean': 0.15576599538326263, 'reward_total_mean': 0.6418274641036987, 'reward_meter_mean': 0.8556537628173828, 'reward_meter_std': 0.34665367007255554, 'reward_count_adherence_mean': 0.949999988079071, 'reward_count_adherence_std': 0.09258200973272324, 'reward_hard_gate_mean': 0.875, 'reward_hard_gate_std': 0.3535533845424652, 'reward_repeat_soft_mean': 0.9829007983207703, 'reward_repeat_soft_std': 0.016816018149256706, 'reward_judge_quality_mean': 0.39374998211860657, 'reward_judge_quality_std': 0.15638209879398346, 'reward_total_composite_mean': 0.6418274641036987, 'reward_total_composite_std': 0.3184812068939209, 'epoch': 0.01} + 4%|▍ | 148/3300 [19:57<6:06:15, 6.97s/it]INFO 04-12 22:34:25 [block_pool.py:378] Successfully reset prefix cache + 5%|▍ | 149/3300 [20:05<6:25:11, 7.33s/it]2026-04-12 22:34:33,260 | INFO | train_grpo_train | metrics_logged mode=train step=149 + {'loss': 0.0005, 'grad_norm': 5.917326927185059, 'learning_rate': 9.551515151515152e-06, 'num_tokens': 289547.0, 'completions/mean_length': 196.125, 'completions/min_length': 170.0, 'completions/max_length': 207.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 196.125, 'completions/min_terminated_length': 170.0, 'completions/max_terminated_length': 207.0, 'rewards/meter/mean': 0.9900409579277039, 'rewards/meter/std': 0.01028269249945879, 'rewards/count_adherence/mean': 0.9791666269302368, 'rewards/count_adherence/std': 0.0589255727827549, 'rewards/hard_gate/mean': 0.875, 'rewards/hard_gate/std': 0.3535533845424652, 'rewards/repeat_soft/mean': 0.9094256162643433, 'rewards/repeat_soft/std': 0.09267400950193405, 'rewards/judge_quality/mean': 0.2212499976158142, 'rewards/judge_quality/std': 0.09433034062385559, 'rewards/total_composite/mean': 0.657077431678772, 'rewards/total_composite/std': 0.2678874731063843, 'reward': 0.657077431678772, 'reward_std': 0.2678874731063843, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.17067110538482666, 'sampling/sampling_logp_difference/max': 1.832137107849121, 'sampling/importance_sampling_ratio/min': 0.16007111966609955, 'sampling/importance_sampling_ratio/mean': 1.0396323204040527, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 2.1036142259836197, 'clip_ratio/low_mean': 0.018518518656492233, 'clip_ratio/low_min': 0.018518518656492233, 'clip_ratio/high_mean': 0.12627116777002811, 'clip_ratio/high_max': 0.12627116777002811, 'clip_ratio/region_mean': 0.14478968642652035, 'reward_total_mean': 0.657077431678772, 'reward_meter_mean': 0.9900409579277039, 'reward_meter_std': 0.01028269249945879, 'reward_count_adherence_mean': 0.9791666269302368, 'reward_count_adherence_std': 0.0589255727827549, 'reward_hard_gate_mean': 0.875, 'reward_hard_gate_std': 0.3535533845424652, 'reward_repeat_soft_mean': 0.9094256162643433, 'reward_repeat_soft_std': 0.09267400950193405, 'reward_judge_quality_mean': 0.2212499976158142, 'reward_judge_quality_std': 0.09433034062385559, 'reward_total_composite_mean': 0.657077431678772, 'reward_total_composite_std': 0.2678874731063843, 'epoch': 0.01} + 5%|▍ | 149/3300 [20:05<6:25:11, 7.33s/it]INFO 04-12 22:34:33 [block_pool.py:378] Successfully reset prefix cache + 5%|▍ | 150/3300 [20:13<6:32:21, 7.47s/it]2026-04-12 22:34:41,057 | INFO | train_grpo_train | metrics_logged mode=train step=150 + {'loss': 0.0307, 'grad_norm': 8.67408275604248, 'learning_rate': 9.54848484848485e-06, 'num_tokens': 291984.0, 'completions/mean_length': 134.625, 'completions/min_length': 114.0, 'completions/max_length': 164.0, 'completions/clipped_ratio': 0.0, 'completions/mean_terminated_length': 134.625, 'completions/min_terminated_length': 114.0, 'completions/max_terminated_length': 164.0, 'rewards/meter/mean': 0.6442027688026428, 'rewards/meter/std': 0.23864556849002838, 'rewards/count_adherence/mean': 0.9249999523162842, 'rewards/count_adherence/std': 0.1035098284482956, 'rewards/hard_gate/mean': 1.0, 'rewards/hard_gate/std': 0.0, 'rewards/repeat_soft/mean': 0.9940277338027954, 'rewards/repeat_soft/std': 0.0051259868778288364, 'rewards/judge_quality/mean': 0.3774999976158142, 'rewards/judge_quality/std': 0.07869470119476318, 'rewards/total_composite/mean': 0.641294002532959, 'rewards/total_composite/std': 0.1019832044839859, 'reward': 0.641294002532959, 'reward_std': 0.1019832044839859, 'frac_reward_zero_std': 0.0, 'sampling/sampling_logp_difference/mean': 0.19296503067016602, 'sampling/sampling_logp_difference/max': 1.9423902034759521, 'sampling/importance_sampling_ratio/min': 0.14336088299751282, 'sampling/importance_sampling_ratio/mean': 1.0387064218521118, 'sampling/importance_sampling_ratio/max': 2.0, 'entropy': 1.8878242373466492, 'clip_ratio/low_mean': 0.12727568112313747, 'clip_ratio/low_min': 0.12727568112313747, 'clip_ratio/high_mean': 0.06800262071192265, 'clip_ratio/high_max': 0.06800262071192265, 'clip_ratio/region_mean': 0.19527830183506012, 'reward_total_mean': 0.641294002532959, 'reward_meter_mean': 0.6442027688026428, 'reward_meter_std': 0.23864556849002838, 'reward_count_adherence_mean': 0.9249999523162842, 'reward_count_adherence_std': 0.1035098284482956, 'reward_hard_gate_mean': 1.0, 'reward_hard_gate_std': 0.0, 'reward_repeat_soft_mean': 0.9940277338027954, 'reward_repeat_soft_std': 0.0051259868778288364, 'reward_judge_quality_mean': 0.3774999976158142, 'reward_judge_quality_std': 0.07869470119476318, 'reward_total_composite_mean': 0.641294002532959, 'reward_total_composite_std': 0.1019832044839859, 'epoch': 0.02} + 5%|▍ | 150/3300 [20:13<6:32:21, 7.47s/it]INFO 04-12 22:34:41 [block_pool.py:378] Successfully reset prefix cache +/root/workspace/Shaer/grpo/.venv/lib/python3.11/site-packages/trl/trainer/grpo_trainer.py:1450: UserWarning: std(): degrees of freedom is <= 0. Correction should be strictly less than the reduction factor (input numel divided by output numel). (Triggered internally at /pytorch/aten/src/ATen/native/ReduceOps.cpp:1839.) + std_rewards = rewards.view(-1, self.num_generations).std(dim=1) + + 0%| | 0/10 [00:00