{"timestamp_utc": "2026-04-13T07:21:02Z", "mode": "train", "global_step": 1, "epoch": 0.00010045203415369161, "loss": 0.1336, "grad_norm": 23.196989059448242, "learning_rate": 1e-05, "num_tokens": 1670.0, "completions/mean_length": 41.75, "completions/min_length": 29.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.75, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.6725290417671204, "rewards/meter/std": 0.40340036153793335, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9957022070884705, "rewards/repeat_soft/std": 0.00409209867939353, "rewards/judge_quality/mean": 0.4762499928474426, "rewards/judge_quality/std": 0.09941796213388443, "rewards/total_composite/mean": 0.5380368828773499, "rewards/total_composite/std": 0.16582772135734558, "reward": 0.5380368828773499, "reward_std": 0.1658277064561844, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22603146731853485, "sampling/sampling_logp_difference/max": 1.5626206398010254, "sampling/importance_sampling_ratio/min": 0.20958609879016876, "sampling/importance_sampling_ratio/mean": 1.013262391090393, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.01923768222332, "clip_ratio/low_mean": 0.10917956382036209, "clip_ratio/low_min": 0.10917956382036209, "clip_ratio/high_mean": 0.12637650407850742, "clip_ratio/high_max": 0.12637650407850742, "clip_ratio/region_mean": 0.23555606789886951, "reward_total_mean": 0.5380368828773499, "reward_meter_mean": 0.6725290417671204, "reward_meter_std": 0.40340036153793335, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9957022070884705, "reward_repeat_soft_std": 0.00409209867939353, "reward_judge_quality_mean": 0.4762499928474426, "reward_judge_quality_std": 0.09941796213388443, "reward_total_composite_mean": 0.5380368828773499, "reward_total_composite_std": 0.16582772135734558} {"timestamp_utc": "2026-04-13T07:21:10Z", "mode": "train", "global_step": 2, "epoch": 0.00020090406830738323, "loss": 0.0546, "grad_norm": 8.660832405090332, "learning_rate": 9.996969696969698e-06, "num_tokens": 4114.0, "completions/mean_length": 141.5, "completions/min_length": 100.0, "completions/max_length": 157.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 141.5, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 157.0, "rewards/meter/mean": 0.6691720485687256, "rewards/meter/std": 0.3117619752883911, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9969235062599182, "rewards/repeat_soft/std": 0.003796192118898034, "rewards/judge_quality/mean": 0.5324999690055847, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.5947020649909973, "rewards/total_composite/std": 0.15181177854537964, "reward": 0.5947020649909973, "reward_std": 0.15181177854537964, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19878226518630981, "sampling/sampling_logp_difference/max": 1.3262004852294922, "sampling/importance_sampling_ratio/min": 0.2654840648174286, "sampling/importance_sampling_ratio/mean": 1.028130292892456, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9687132090330124, "clip_ratio/low_mean": 0.08277223259210587, "clip_ratio/low_min": 0.08277223259210587, "clip_ratio/high_mean": 0.09919708035886288, "clip_ratio/high_max": 0.09919708035886288, "clip_ratio/region_mean": 0.18196931295096874, "reward_total_mean": 0.5947020649909973, "reward_meter_mean": 0.6691720485687256, "reward_meter_std": 0.3117619752883911, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9969235062599182, "reward_repeat_soft_std": 0.003796192118898034, "reward_judge_quality_mean": 0.5324999690055847, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.5947020649909973, "reward_total_composite_std": 0.15181177854537964} {"timestamp_utc": "2026-04-13T07:21:18Z", "mode": "train", "global_step": 3, "epoch": 0.00030135610246107485, "loss": 0.1089, "grad_norm": 23.679033279418945, "learning_rate": 9.993939393939395e-06, "num_tokens": 5877.0, "completions/mean_length": 43.375, "completions/min_length": 31.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.375, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.44267401099205017, "rewards/meter/std": 0.36050936579704285, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.992233157157898, "rewards/repeat_soft/std": 0.012951970100402832, "rewards/judge_quality/mean": 0.6225000023841858, "rewards/judge_quality/std": 0.24656209349632263, "rewards/total_composite/mean": 0.5438969135284424, "rewards/total_composite/std": 0.21351033449172974, "reward": 0.5438969135284424, "reward_std": 0.21351033449172974, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2622106671333313, "sampling/sampling_logp_difference/max": 2.150146245956421, "sampling/importance_sampling_ratio/min": 0.11646712571382523, "sampling/importance_sampling_ratio/mean": 1.0123623609542847, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2060932368040085, "clip_ratio/low_mean": 0.13107881788164377, "clip_ratio/low_min": 0.13107881788164377, "clip_ratio/high_mean": 0.08155973814427853, "clip_ratio/high_max": 0.08155973814427853, "clip_ratio/region_mean": 0.2126385560259223, "reward_total_mean": 0.5438969135284424, "reward_meter_mean": 0.44267401099205017, "reward_meter_std": 0.36050936579704285, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.992233157157898, "reward_repeat_soft_std": 0.012951970100402832, "reward_judge_quality_mean": 0.6225000023841858, "reward_judge_quality_std": 0.24656209349632263, "reward_total_composite_mean": 0.5438969135284424, "reward_total_composite_std": 0.21351033449172974} {"timestamp_utc": "2026-04-13T07:21:25Z", "mode": "train", "global_step": 4, "epoch": 0.00040180813661476645, "loss": 0.1969, "grad_norm": 17.794963836669922, "learning_rate": 9.990909090909093e-06, "num_tokens": 7527.0, "completions/mean_length": 49.25, "completions/min_length": 31.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.25, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.514047384262085, "rewards/meter/std": 0.4770835041999817, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9997227191925049, "rewards/repeat_soft/std": 0.0006712718168273568, "rewards/judge_quality/mean": 0.7325000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.6147944927215576, "rewards/total_composite/std": 0.264877587556839, "reward": 0.6147944927215576, "reward_std": 0.264877587556839, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2241060584783554, "sampling/sampling_logp_difference/max": 3.2926888465881348, "sampling/importance_sampling_ratio/min": 0.03715381398797035, "sampling/importance_sampling_ratio/mean": 1.0513813495635986, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8340110927820206, "clip_ratio/low_mean": 0.08712000586092472, "clip_ratio/low_min": 0.08712000586092472, "clip_ratio/high_mean": 0.10127950087189674, "clip_ratio/high_max": 0.10127950087189674, "clip_ratio/region_mean": 0.18839950673282146, "reward_total_mean": 0.6147944927215576, "reward_meter_mean": 0.514047384262085, "reward_meter_std": 0.4770835041999817, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9997227191925049, "reward_repeat_soft_std": 0.0006712718168273568, "reward_judge_quality_mean": 0.7325000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.6147944927215576, "reward_total_composite_std": 0.264877587556839} {"timestamp_utc": "2026-04-13T07:21:33Z", "mode": "train", "global_step": 5, "epoch": 0.0005022601707684581, "loss": 0.1922, "grad_norm": 8.504620552062988, "learning_rate": 9.987878787878788e-06, "num_tokens": 9871.0, "completions/mean_length": 116.0, "completions/min_length": 77.0, "completions/max_length": 140.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 116.0, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 140.0, "rewards/meter/mean": 0.400786817073822, "rewards/meter/std": 0.38270607590675354, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9964909553527832, "rewards/repeat_soft/std": 0.004073499236255884, "rewards/judge_quality/mean": 0.7074999809265137, "rewards/judge_quality/std": 0.247487410902977, "rewards/total_composite/mean": 0.526107668876648, "rewards/total_composite/std": 0.1974838376045227, "reward": 0.526107668876648, "reward_std": 0.1974838227033615, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17819346487522125, "sampling/sampling_logp_difference/max": 1.6136455535888672, "sampling/importance_sampling_ratio/min": 0.19916023313999176, "sampling/importance_sampling_ratio/mean": 1.0168025493621826, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.27456446737051, "clip_ratio/low_mean": 0.10275345481932163, "clip_ratio/low_min": 0.10275345481932163, "clip_ratio/high_mean": 0.0539776710793376, "clip_ratio/high_max": 0.0539776710793376, "clip_ratio/region_mean": 0.15673112589865923, "reward_total_mean": 0.526107668876648, "reward_meter_mean": 0.400786817073822, "reward_meter_std": 0.38270607590675354, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9964909553527832, "reward_repeat_soft_std": 0.004073499236255884, "reward_judge_quality_mean": 0.7074999809265137, "reward_judge_quality_std": 0.247487410902977, "reward_total_composite_mean": 0.526107668876648, "reward_total_composite_std": 0.1974838376045227} {"timestamp_utc": "2026-04-13T07:21:41Z", "mode": "train", "global_step": 6, "epoch": 0.0006027122049221497, "loss": 0.0954, "grad_norm": 11.160085678100586, "learning_rate": 9.984848484848485e-06, "num_tokens": 12514.0, "completions/mean_length": 131.375, "completions/min_length": 108.0, "completions/max_length": 158.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.375, "completions/min_terminated_length": 108.0, "completions/max_terminated_length": 158.0, "rewards/meter/mean": 0.8979466557502747, "rewards/meter/std": 0.27314630150794983, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9976483583450317, "rewards/repeat_soft/std": 0.0014516275841742754, "rewards/judge_quality/mean": 0.6137499809265137, "rewards/judge_quality/std": 0.2650572657585144, "rewards/total_composite/mean": 0.7199420928955078, "rewards/total_composite/std": 0.20595994591712952, "reward": 0.7199420928955078, "reward_std": 0.20595994591712952, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20203472673892975, "sampling/sampling_logp_difference/max": 1.973508358001709, "sampling/importance_sampling_ratio/min": 0.13896843791007996, "sampling/importance_sampling_ratio/mean": 1.0401043891906738, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2473559081554413, "clip_ratio/low_mean": 0.0936892218887806, "clip_ratio/low_min": 0.0936892218887806, "clip_ratio/high_mean": 0.10637771897017956, "clip_ratio/high_max": 0.10637771897017956, "clip_ratio/region_mean": 0.20006694085896015, "reward_total_mean": 0.7199420928955078, "reward_meter_mean": 0.8979466557502747, "reward_meter_std": 0.27314630150794983, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9976483583450317, "reward_repeat_soft_std": 0.0014516275841742754, "reward_judge_quality_mean": 0.6137499809265137, "reward_judge_quality_std": 0.2650572657585144, "reward_total_composite_mean": 0.7199420928955078, "reward_total_composite_std": 0.20595994591712952} {"timestamp_utc": "2026-04-13T07:21:48Z", "mode": "train", "global_step": 7, "epoch": 0.0007031642390758413, "loss": -0.0688, "grad_norm": 11.945133209228516, "learning_rate": 9.981818181818183e-06, "num_tokens": 14844.0, "completions/mean_length": 107.25, "completions/min_length": 81.0, "completions/max_length": 149.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.25, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 149.0, "rewards/meter/mean": 0.3544602692127228, "rewards/meter/std": 0.3453842103481293, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.999114990234375, "rewards/repeat_soft/std": 0.0010229938197880983, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.372326135635376, "rewards/total_composite/std": 0.16378259658813477, "reward": 0.372326135635376, "reward_std": 0.16378259658813477, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2678569257259369, "sampling/sampling_logp_difference/max": 1.720468521118164, "sampling/importance_sampling_ratio/min": 0.17898227274417877, "sampling/importance_sampling_ratio/mean": 1.018048644065857, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.666846424341202, "clip_ratio/low_mean": 0.05837517976760864, "clip_ratio/low_min": 0.05837517976760864, "clip_ratio/high_mean": 0.2030274923890829, "clip_ratio/high_max": 0.2030274923890829, "clip_ratio/region_mean": 0.26140267215669155, "reward_total_mean": 0.372326135635376, "reward_meter_mean": 0.3544602692127228, "reward_meter_std": 0.3453842103481293, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.999114990234375, "reward_repeat_soft_std": 0.0010229938197880983, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.372326135635376, "reward_total_composite_std": 0.16378259658813477} {"timestamp_utc": "2026-04-13T07:21:55Z", "mode": "train", "global_step": 8, "epoch": 0.0008036162732295329, "loss": 0.0948, "grad_norm": 24.52655029296875, "learning_rate": 9.97878787878788e-06, "num_tokens": 16305.0, "completions/mean_length": 26.625, "completions/min_length": 18.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.625, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.7332162261009216, "rewards/meter/std": 0.33431848883628845, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9578598141670227, "rewards/repeat_soft/std": 0.01113725546747446, "rewards/judge_quality/mean": 0.7362500429153442, "rewards/judge_quality/std": 0.2620762288570404, "rewards/total_composite/mean": 0.7266648411750793, "rewards/total_composite/std": 0.22824539244174957, "reward": 0.7266648411750793, "reward_std": 0.22824537754058838, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19474588334560394, "sampling/sampling_logp_difference/max": 3.4640512466430664, "sampling/importance_sampling_ratio/min": 0.031302690505981445, "sampling/importance_sampling_ratio/mean": 1.0509170293807983, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5002107545733452, "clip_ratio/low_mean": 0.07552917674183846, "clip_ratio/low_min": 0.07552917674183846, "clip_ratio/high_mean": 0.0975095797330141, "clip_ratio/high_max": 0.0975095797330141, "clip_ratio/region_mean": 0.17303875647485256, "reward_total_mean": 0.7266648411750793, "reward_meter_mean": 0.7332162261009216, "reward_meter_std": 0.33431848883628845, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9578598141670227, "reward_repeat_soft_std": 0.01113725546747446, "reward_judge_quality_mean": 0.7362500429153442, "reward_judge_quality_std": 0.2620762288570404, "reward_total_composite_mean": 0.7266648411750793, "reward_total_composite_std": 0.22824539244174957} {"timestamp_utc": "2026-04-13T07:22:03Z", "mode": "train", "global_step": 9, "epoch": 0.0009040683073832245, "loss": 0.169, "grad_norm": 8.872598648071289, "learning_rate": 9.975757575757577e-06, "num_tokens": 19099.0, "completions/mean_length": 143.25, "completions/min_length": 92.0, "completions/max_length": 192.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 143.25, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 192.0, "rewards/meter/mean": 0.8696469068527222, "rewards/meter/std": 0.18565556406974792, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9950947761535645, "rewards/repeat_soft/std": 0.006243214942514896, "rewards/judge_quality/mean": 0.5112500190734863, "rewards/judge_quality/std": 0.18216457962989807, "rewards/total_composite/mean": 0.6252528429031372, "rewards/total_composite/std": 0.0826655700802803, "reward": 0.6252528429031372, "reward_std": 0.0826655700802803, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1975843459367752, "sampling/sampling_logp_difference/max": 3.2445778846740723, "sampling/importance_sampling_ratio/min": 0.03898501768708229, "sampling/importance_sampling_ratio/mean": 1.0382981300354004, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.174727141857147, "clip_ratio/low_mean": 0.16696189157664776, "clip_ratio/low_min": 0.16696189157664776, "clip_ratio/high_mean": 0.02835051529109478, "clip_ratio/high_max": 0.02835051529109478, "clip_ratio/region_mean": 0.19531240686774254, "reward_total_mean": 0.6252528429031372, "reward_meter_mean": 0.8696469068527222, "reward_meter_std": 0.18565556406974792, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9950947761535645, "reward_repeat_soft_std": 0.006243214942514896, "reward_judge_quality_mean": 0.5112500190734863, "reward_judge_quality_std": 0.18216457962989807, "reward_total_composite_mean": 0.6252528429031372, "reward_total_composite_std": 0.0826655700802803} {"timestamp_utc": "2026-04-13T07:22:10Z", "mode": "train", "global_step": 10, "epoch": 0.0010045203415369162, "loss": -0.0406, "grad_norm": 18.636837005615234, "learning_rate": 9.972727272727274e-06, "num_tokens": 20677.0, "completions/mean_length": 35.25, "completions/min_length": 24.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.25, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.878533124923706, "rewards/meter/std": 0.1565551459789276, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9728596210479736, "rewards/repeat_soft/std": 0.036961715668439865, "rewards/judge_quality/mean": 0.8025000095367432, "rewards/judge_quality/std": 0.21756774187088013, "rewards/total_composite/mean": 0.6499476432800293, "rewards/total_composite/std": 0.41783440113067627, "reward": 0.6499476432800293, "reward_std": 0.41783440113067627, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21180443465709686, "sampling/sampling_logp_difference/max": 2.064453125, "sampling/importance_sampling_ratio/min": 0.12688764929771423, "sampling/importance_sampling_ratio/mean": 0.9985918402671814, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1395761594176292, "clip_ratio/low_mean": 0.08045504428446293, "clip_ratio/low_min": 0.08045504428446293, "clip_ratio/high_mean": 0.10815705172717571, "clip_ratio/high_max": 0.10815705172717571, "clip_ratio/region_mean": 0.18861209601163864, "reward_total_mean": 0.6499476432800293, "reward_meter_mean": 0.878533124923706, "reward_meter_std": 0.1565551459789276, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9728596210479736, "reward_repeat_soft_std": 0.036961715668439865, "reward_judge_quality_mean": 0.8025000095367432, "reward_judge_quality_std": 0.21756774187088013, "reward_total_composite_mean": 0.6499476432800293, "reward_total_composite_std": 0.41783440113067627} {"timestamp_utc": "2026-04-13T07:22:16Z", "mode": "train", "global_step": 11, "epoch": 0.0011049723756906078, "loss": 0.0526, "grad_norm": 22.84332847595215, "learning_rate": 9.96969696969697e-06, "num_tokens": 22159.0, "completions/mean_length": 35.25, "completions/min_length": 31.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.25, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.42980778217315674, "rewards/meter/std": 0.4602866768836975, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9971010684967041, "rewards/repeat_soft/std": 0.004350362345576286, "rewards/judge_quality/mean": 0.6025000214576721, "rewards/judge_quality/std": 0.1976107507944107, "rewards/total_composite/mean": 0.517084002494812, "rewards/total_composite/std": 0.2049219310283661, "reward": 0.517084002494812, "reward_std": 0.2049219161272049, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.25400835275650024, "sampling/sampling_logp_difference/max": 1.6263856887817383, "sampling/importance_sampling_ratio/min": 0.19663900136947632, "sampling/importance_sampling_ratio/mean": 1.044650673866272, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.648899346590042, "clip_ratio/low_mean": 0.11001225002110004, "clip_ratio/low_min": 0.11001225002110004, "clip_ratio/high_mean": 0.0788961062207818, "clip_ratio/high_max": 0.0788961062207818, "clip_ratio/region_mean": 0.18890835624188185, "reward_total_mean": 0.517084002494812, "reward_meter_mean": 0.42980778217315674, "reward_meter_std": 0.4602866768836975, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9971010684967041, "reward_repeat_soft_std": 0.004350362345576286, "reward_judge_quality_mean": 0.6025000214576721, "reward_judge_quality_std": 0.1976107507944107, "reward_total_composite_mean": 0.517084002494812, "reward_total_composite_std": 0.2049219310283661} {"timestamp_utc": "2026-04-13T07:22:23Z", "mode": "train", "global_step": 12, "epoch": 0.0012054244098442994, "loss": 0.1019, "grad_norm": 23.60201072692871, "learning_rate": 9.966666666666667e-06, "num_tokens": 23601.0, "completions/mean_length": 20.25, "completions/min_length": 15.0, "completions/max_length": 27.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 20.25, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.5795812606811523, "rewards/meter/std": 0.4281229078769684, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9527608156204224, "rewards/repeat_soft/std": 0.02547159232199192, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.5066647529602051, "rewards/total_composite/std": 0.11989112198352814, "reward": 0.5066647529602051, "reward_std": 0.11989112198352814, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21970289945602417, "sampling/sampling_logp_difference/max": 0.9635438919067383, "sampling/importance_sampling_ratio/min": 0.3823465406894684, "sampling/importance_sampling_ratio/mean": 1.056498408317566, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.821111336350441, "clip_ratio/low_mean": 0.10659844242036343, "clip_ratio/low_min": 0.10659844242036343, "clip_ratio/high_mean": 0.08916170708835125, "clip_ratio/high_max": 0.08916170708835125, "clip_ratio/region_mean": 0.19576014950871468, "reward_total_mean": 0.5066647529602051, "reward_meter_mean": 0.5795812606811523, "reward_meter_std": 0.4281229078769684, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9527608156204224, "reward_repeat_soft_std": 0.02547159232199192, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.5066647529602051, "reward_total_composite_std": 0.11989112198352814} {"timestamp_utc": "2026-04-13T07:22:29Z", "mode": "train", "global_step": 13, "epoch": 0.001305876443997991, "loss": 0.0329, "grad_norm": 17.67831039428711, "learning_rate": 9.963636363636364e-06, "num_tokens": 25165.0, "completions/mean_length": 38.5, "completions/min_length": 24.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.5, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.5992826819419861, "rewards/meter/std": 0.3339884281158447, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9907096028327942, "rewards/repeat_soft/std": 0.014991275034844875, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.5917410850524902, "rewards/total_composite/std": 0.22081266343593597, "reward": 0.5917410850524902, "reward_std": 0.22081266343593597, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24050824344158173, "sampling/sampling_logp_difference/max": 1.6661157608032227, "sampling/importance_sampling_ratio/min": 0.18897968530654907, "sampling/importance_sampling_ratio/mean": 0.9785161018371582, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0621970295906067, "clip_ratio/low_mean": 0.14360823296010494, "clip_ratio/low_min": 0.14360823296010494, "clip_ratio/high_mean": 0.09686947427690029, "clip_ratio/high_max": 0.09686947427690029, "clip_ratio/region_mean": 0.24047770723700523, "reward_total_mean": 0.5917410850524902, "reward_meter_mean": 0.5992826819419861, "reward_meter_std": 0.3339884281158447, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9907096028327942, "reward_repeat_soft_std": 0.014991275034844875, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.5917410850524902, "reward_total_composite_std": 0.22081266343593597} {"timestamp_utc": "2026-04-13T07:22:38Z", "mode": "train", "global_step": 14, "epoch": 0.0014063284781516826, "loss": -0.0512, "grad_norm": 11.103429794311523, "learning_rate": 9.960606060606062e-06, "num_tokens": 27784.0, "completions/mean_length": 133.375, "completions/min_length": 89.0, "completions/max_length": 209.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 133.375, "completions/min_terminated_length": 89.0, "completions/max_terminated_length": 209.0, "rewards/meter/mean": 0.7462164163589478, "rewards/meter/std": 0.40208691358566284, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9947032928466797, "rewards/repeat_soft/std": 0.006884792819619179, "rewards/judge_quality/mean": 0.5774999856948853, "rewards/judge_quality/std": 0.1527603417634964, "rewards/total_composite/mean": 0.6377272009849548, "rewards/total_composite/std": 0.19227305054664612, "reward": 0.6377272009849548, "reward_std": 0.19227305054664612, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19112931191921234, "sampling/sampling_logp_difference/max": 1.8332371711730957, "sampling/importance_sampling_ratio/min": 0.18594108521938324, "sampling/importance_sampling_ratio/mean": 1.0424418449401855, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.17301943898201, "clip_ratio/low_mean": 0.0681159496307373, "clip_ratio/low_min": 0.0681159496307373, "clip_ratio/high_mean": 0.11231567524373531, "clip_ratio/high_max": 0.11231567524373531, "clip_ratio/region_mean": 0.18043162487447262, "reward_total_mean": 0.6377272009849548, "reward_meter_mean": 0.7462164163589478, "reward_meter_std": 0.40208691358566284, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9947032928466797, "reward_repeat_soft_std": 0.006884792819619179, "reward_judge_quality_mean": 0.5774999856948853, "reward_judge_quality_std": 0.1527603417634964, "reward_total_composite_mean": 0.6377272009849548, "reward_total_composite_std": 0.19227305054664612} {"timestamp_utc": "2026-04-13T07:22:45Z", "mode": "train", "global_step": 15, "epoch": 0.0015067805123053742, "loss": 0.141, "grad_norm": 17.71084976196289, "learning_rate": 9.957575757575757e-06, "num_tokens": 29510.0, "completions/mean_length": 50.75, "completions/min_length": 33.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.75, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.3207266628742218, "rewards/meter/std": 0.35546308755874634, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9936358332633972, "rewards/repeat_soft/std": 0.010752102360129356, "rewards/judge_quality/mean": 0.5637500286102295, "rewards/judge_quality/std": 0.22012579441070557, "rewards/total_composite/mean": 0.4819847345352173, "rewards/total_composite/std": 0.1911517083644867, "reward": 0.4819847345352173, "reward_std": 0.1911516934633255, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2281627207994461, "sampling/sampling_logp_difference/max": 2.519559860229492, "sampling/importance_sampling_ratio/min": 0.08049502968788147, "sampling/importance_sampling_ratio/mean": 1.0143530368804932, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7476065307855606, "clip_ratio/low_mean": 0.17134212143719196, "clip_ratio/low_min": 0.17134212143719196, "clip_ratio/high_mean": 0.0547595527023077, "clip_ratio/high_max": 0.0547595527023077, "clip_ratio/region_mean": 0.22610167413949966, "reward_total_mean": 0.4819847345352173, "reward_meter_mean": 0.3207266628742218, "reward_meter_std": 0.35546308755874634, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9936358332633972, "reward_repeat_soft_std": 0.010752102360129356, "reward_judge_quality_mean": 0.5637500286102295, "reward_judge_quality_std": 0.22012579441070557, "reward_total_composite_mean": 0.4819847345352173, "reward_total_composite_std": 0.1911517083644867} {"timestamp_utc": "2026-04-13T07:22:52Z", "mode": "train", "global_step": 16, "epoch": 0.0016072325464590658, "loss": -0.0027, "grad_norm": 17.285341262817383, "learning_rate": 9.954545454545456e-06, "num_tokens": 31067.0, "completions/mean_length": 43.625, "completions/min_length": 31.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.625, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.4554264545440674, "rewards/meter/std": 0.4676746129989624, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9939950108528137, "rewards/repeat_soft/std": 0.004853702150285244, "rewards/judge_quality/mean": 0.6225000023841858, "rewards/judge_quality/std": 0.24656209349632263, "rewards/total_composite/mean": 0.5386902093887329, "rewards/total_composite/std": 0.21134473383426666, "reward": 0.5386902093887329, "reward_std": 0.21134471893310547, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23214435577392578, "sampling/sampling_logp_difference/max": 1.6480565071105957, "sampling/importance_sampling_ratio/min": 0.1924235224723816, "sampling/importance_sampling_ratio/mean": 1.0233776569366455, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.098197489976883, "clip_ratio/low_mean": 0.10589986108243465, "clip_ratio/low_min": 0.10589986108243465, "clip_ratio/high_mean": 0.10519339889287949, "clip_ratio/high_max": 0.10519339889287949, "clip_ratio/region_mean": 0.21109325997531414, "reward_total_mean": 0.5386902093887329, "reward_meter_mean": 0.4554264545440674, "reward_meter_std": 0.4676746129989624, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9939950108528137, "reward_repeat_soft_std": 0.004853702150285244, "reward_judge_quality_mean": 0.6225000023841858, "reward_judge_quality_std": 0.24656209349632263, "reward_total_composite_mean": 0.5386902093887329, "reward_total_composite_std": 0.21134473383426666} {"timestamp_utc": "2026-04-13T07:22:59Z", "mode": "train", "global_step": 17, "epoch": 0.0017076845806127574, "loss": -0.0109, "grad_norm": 17.762605667114258, "learning_rate": 9.951515151515152e-06, "num_tokens": 32694.0, "completions/mean_length": 44.375, "completions/min_length": 32.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.375, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.4266776740550995, "rewards/meter/std": 0.3314104378223419, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9950998425483704, "rewards/repeat_soft/std": 0.007583285681903362, "rewards/judge_quality/mean": 0.4424999952316284, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.47071021795272827, "rewards/total_composite/std": 0.09301955997943878, "reward": 0.47071021795272827, "reward_std": 0.09301956743001938, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2339089810848236, "sampling/sampling_logp_difference/max": 1.5349130630493164, "sampling/importance_sampling_ratio/min": 0.2154744267463684, "sampling/importance_sampling_ratio/mean": 1.0189992189407349, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1940976977348328, "clip_ratio/low_mean": 0.1057262234389782, "clip_ratio/low_min": 0.1057262234389782, "clip_ratio/high_mean": 0.07310517132282257, "clip_ratio/high_max": 0.07310517132282257, "clip_ratio/region_mean": 0.17883139476180077, "reward_total_mean": 0.47071021795272827, "reward_meter_mean": 0.4266776740550995, "reward_meter_std": 0.3314104378223419, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9950998425483704, "reward_repeat_soft_std": 0.007583285681903362, "reward_judge_quality_mean": 0.4424999952316284, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.47071021795272827, "reward_total_composite_std": 0.09301955997943878} {"timestamp_utc": "2026-04-13T07:23:05Z", "mode": "train", "global_step": 18, "epoch": 0.001808136614766449, "loss": -0.1495, "grad_norm": 15.353915214538574, "learning_rate": 9.948484848484849e-06, "num_tokens": 34119.0, "completions/mean_length": 29.125, "completions/min_length": 5.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.125, "completions/min_terminated_length": 5.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.7113527059555054, "rewards/meter/std": 0.41699108481407166, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9671874642372131, "rewards/repeat_soft/std": 0.013258260674774647, "rewards/judge_quality/mean": 0.5699999928474426, "rewards/judge_quality/std": 0.32732686400413513, "rewards/total_composite/mean": 0.5885436534881592, "rewards/total_composite/std": 0.2998615503311157, "reward": 0.5885436534881592, "reward_std": 0.29986152052879333, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23092404007911682, "sampling/sampling_logp_difference/max": 1.4778499603271484, "sampling/importance_sampling_ratio/min": 0.22812765836715698, "sampling/importance_sampling_ratio/mean": 1.0105531215667725, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8655916303396225, "clip_ratio/low_mean": 0.05056818202137947, "clip_ratio/low_min": 0.05056818202137947, "clip_ratio/high_mean": 0.15987520338967443, "clip_ratio/high_max": 0.15987520338967443, "clip_ratio/region_mean": 0.2104433854110539, "reward_total_mean": 0.5885436534881592, "reward_meter_mean": 0.7113527059555054, "reward_meter_std": 0.41699108481407166, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9671874642372131, "reward_repeat_soft_std": 0.013258260674774647, "reward_judge_quality_mean": 0.5699999928474426, "reward_judge_quality_std": 0.32732686400413513, "reward_total_composite_mean": 0.5885436534881592, "reward_total_composite_std": 0.2998615503311157} {"timestamp_utc": "2026-04-13T07:23:13Z", "mode": "train", "global_step": 19, "epoch": 0.0019085886489201406, "loss": -0.0096, "grad_norm": 14.933284759521484, "learning_rate": 9.945454545454546e-06, "num_tokens": 35830.0, "completions/mean_length": 58.875, "completions/min_length": 43.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.875, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.7577532529830933, "rewards/meter/std": 0.42325934767723083, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9988213777542114, "rewards/repeat_soft/std": 0.0029925131238996983, "rewards/judge_quality/mean": 0.7400000095367432, "rewards/judge_quality/std": 0.24859607219696045, "rewards/total_composite/mean": 0.7470488548278809, "rewards/total_composite/std": 0.2544901967048645, "reward": 0.7470488548278809, "reward_std": 0.2544901967048645, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17060860991477966, "sampling/sampling_logp_difference/max": 1.4674263000488281, "sampling/importance_sampling_ratio/min": 0.23051801323890686, "sampling/importance_sampling_ratio/mean": 1.0229806900024414, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2996538653969765, "clip_ratio/low_mean": 0.08071060664951801, "clip_ratio/low_min": 0.08071060664951801, "clip_ratio/high_mean": 0.09031699690967798, "clip_ratio/high_max": 0.09031699690967798, "clip_ratio/region_mean": 0.171027603559196, "reward_total_mean": 0.7470488548278809, "reward_meter_mean": 0.7577532529830933, "reward_meter_std": 0.42325934767723083, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9988213777542114, "reward_repeat_soft_std": 0.0029925131238996983, "reward_judge_quality_mean": 0.7400000095367432, "reward_judge_quality_std": 0.24859607219696045, "reward_total_composite_mean": 0.7470488548278809, "reward_total_composite_std": 0.2544901967048645} {"timestamp_utc": "2026-04-13T07:23:20Z", "mode": "train", "global_step": 20, "epoch": 0.0020090406830738324, "loss": 0.0622, "grad_norm": 13.815991401672363, "learning_rate": 9.942424242424244e-06, "num_tokens": 37941.0, "completions/mean_length": 76.875, "completions/min_length": 56.0, "completions/max_length": 134.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.875, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 134.0, "rewards/meter/mean": 0.4842698574066162, "rewards/meter/std": 0.3375895917415619, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.15430334210395813, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9973297119140625, "rewards/repeat_soft/std": 0.002192606683820486, "rewards/judge_quality/mean": 0.5612500309944153, "rewards/judge_quality/std": 0.18946824967861176, "rewards/total_composite/mean": 0.36717891693115234, "rewards/total_composite/std": 0.2395496368408203, "reward": 0.36717891693115234, "reward_std": 0.23954962193965912, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2359773963689804, "sampling/sampling_logp_difference/max": 1.638615608215332, "sampling/importance_sampling_ratio/min": 0.19424878060817719, "sampling/importance_sampling_ratio/mean": 1.0300812721252441, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.34169939160347, "clip_ratio/low_mean": 0.05780553072690964, "clip_ratio/low_min": 0.05780553072690964, "clip_ratio/high_mean": 0.18343774788081646, "clip_ratio/high_max": 0.18343774788081646, "clip_ratio/region_mean": 0.2412432786077261, "reward_total_mean": 0.36717891693115234, "reward_meter_mean": 0.4842698574066162, "reward_meter_std": 0.3375895917415619, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.15430334210395813, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9973297119140625, "reward_repeat_soft_std": 0.002192606683820486, "reward_judge_quality_mean": 0.5612500309944153, "reward_judge_quality_std": 0.18946824967861176, "reward_total_composite_mean": 0.36717891693115234, "reward_total_composite_std": 0.2395496368408203} {"timestamp_utc": "2026-04-13T07:23:27Z", "mode": "train", "global_step": 21, "epoch": 0.002109492717227524, "loss": 0.0865, "grad_norm": 32.94882583618164, "learning_rate": 9.939393939393939e-06, "num_tokens": 39483.0, "completions/mean_length": 33.75, "completions/min_length": 26.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.75, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.2751612365245819, "rewards/meter/std": 0.3222569227218628, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9974137544631958, "rewards/repeat_soft/std": 0.007314986549317837, "rewards/judge_quality/mean": 0.6737500429153442, "rewards/judge_quality/std": 0.263435423374176, "rewards/total_composite/mean": 0.39775148034095764, "rewards/total_composite/std": 0.20206686854362488, "reward": 0.39775148034095764, "reward_std": 0.20206685364246368, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2578023374080658, "sampling/sampling_logp_difference/max": 2.409052848815918, "sampling/importance_sampling_ratio/min": 0.0899004116654396, "sampling/importance_sampling_ratio/mean": 1.0084861516952515, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.245336264371872, "clip_ratio/low_mean": 0.10033134557306767, "clip_ratio/low_min": 0.10033134557306767, "clip_ratio/high_mean": 0.0975898988544941, "clip_ratio/high_max": 0.0975898988544941, "clip_ratio/region_mean": 0.19792124442756176, "reward_total_mean": 0.39775148034095764, "reward_meter_mean": 0.2751612365245819, "reward_meter_std": 0.3222569227218628, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9974137544631958, "reward_repeat_soft_std": 0.007314986549317837, "reward_judge_quality_mean": 0.6737500429153442, "reward_judge_quality_std": 0.263435423374176, "reward_total_composite_mean": 0.39775148034095764, "reward_total_composite_std": 0.20206686854362488} {"timestamp_utc": "2026-04-13T07:23:34Z", "mode": "train", "global_step": 22, "epoch": 0.0022099447513812156, "loss": 0.261, "grad_norm": 14.979035377502441, "learning_rate": 9.936363636363638e-06, "num_tokens": 41219.0, "completions/mean_length": 58.0, "completions/min_length": 37.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.8595972657203674, "rewards/meter/std": 0.2375323474407196, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9979074001312256, "rewards/repeat_soft/std": 0.0036863330751657486, "rewards/judge_quality/mean": 0.5325000286102295, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.6432663798332214, "rewards/total_composite/std": 0.14628660678863525, "reward": 0.6432663798332214, "reward_std": 0.14628660678863525, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19536568224430084, "sampling/sampling_logp_difference/max": 1.2948904037475586, "sampling/importance_sampling_ratio/min": 0.2739278972148895, "sampling/importance_sampling_ratio/mean": 1.0419034957885742, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9579142928123474, "clip_ratio/low_mean": 0.08774491120129824, "clip_ratio/low_min": 0.08774491120129824, "clip_ratio/high_mean": 0.06056361272931099, "clip_ratio/high_max": 0.06056361272931099, "clip_ratio/region_mean": 0.14830852393060923, "reward_total_mean": 0.6432663798332214, "reward_meter_mean": 0.8595972657203674, "reward_meter_std": 0.2375323474407196, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9979074001312256, "reward_repeat_soft_std": 0.0036863330751657486, "reward_judge_quality_mean": 0.5325000286102295, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.6432663798332214, "reward_total_composite_std": 0.14628660678863525} {"timestamp_utc": "2026-04-13T07:23:41Z", "mode": "train", "global_step": 23, "epoch": 0.0023103967855349072, "loss": 0.1544, "grad_norm": 17.582691192626953, "learning_rate": 9.933333333333334e-06, "num_tokens": 42894.0, "completions/mean_length": 49.375, "completions/min_length": 39.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.375, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.4682067632675171, "rewards/meter/std": 0.39334917068481445, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9981834888458252, "rewards/repeat_soft/std": 0.0032929633744060993, "rewards/judge_quality/mean": 0.59375, "rewards/judge_quality/std": 0.22965426743030548, "rewards/total_composite/mean": 0.5504657626152039, "rewards/total_composite/std": 0.20627839863300323, "reward": 0.5504657626152039, "reward_std": 0.20627839863300323, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1894393414258957, "sampling/sampling_logp_difference/max": 2.1364288330078125, "sampling/importance_sampling_ratio/min": 0.11807575821876526, "sampling/importance_sampling_ratio/mean": 1.0336065292358398, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6671224385499954, "clip_ratio/low_mean": 0.10553514678031206, "clip_ratio/low_min": 0.10553514678031206, "clip_ratio/high_mean": 0.05519563052803278, "clip_ratio/high_max": 0.05519563052803278, "clip_ratio/region_mean": 0.16073077730834484, "reward_total_mean": 0.5504657626152039, "reward_meter_mean": 0.4682067632675171, "reward_meter_std": 0.39334917068481445, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9981834888458252, "reward_repeat_soft_std": 0.0032929633744060993, "reward_judge_quality_mean": 0.59375, "reward_judge_quality_std": 0.22965426743030548, "reward_total_composite_mean": 0.5504657626152039, "reward_total_composite_std": 0.20627839863300323} {"timestamp_utc": "2026-04-13T07:23:49Z", "mode": "train", "global_step": 24, "epoch": 0.002410848819688599, "loss": 0.0501, "grad_norm": 13.180197715759277, "learning_rate": 9.930303030303031e-06, "num_tokens": 44607.0, "completions/mean_length": 62.125, "completions/min_length": 46.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.125, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.21943849325180054, "rewards/meter/std": 0.14068876206874847, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9984592199325562, "rewards/repeat_soft/std": 0.00178744294680655, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.4418691396713257, "rewards/total_composite/std": 0.07863625138998032, "reward": 0.4418691396713257, "reward_std": 0.07863625138998032, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18320536613464355, "sampling/sampling_logp_difference/max": 1.4752097129821777, "sampling/importance_sampling_ratio/min": 0.3220268189907074, "sampling/importance_sampling_ratio/mean": 1.0245449542999268, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4479553326964378, "clip_ratio/low_mean": 0.1257912116125226, "clip_ratio/low_min": 0.1257912116125226, "clip_ratio/high_mean": 0.07774893380701542, "clip_ratio/high_max": 0.07774893380701542, "clip_ratio/region_mean": 0.20354014541953802, "reward_total_mean": 0.4418691396713257, "reward_meter_mean": 0.21943849325180054, "reward_meter_std": 0.14068876206874847, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9984592199325562, "reward_repeat_soft_std": 0.00178744294680655, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.4418691396713257, "reward_total_composite_std": 0.07863625138998032} {"timestamp_utc": "2026-04-13T07:23:58Z", "mode": "train", "global_step": 25, "epoch": 0.0025113008538422904, "loss": 0.0264, "grad_norm": 8.850423812866211, "learning_rate": 9.927272727272728e-06, "num_tokens": 47239.0, "completions/mean_length": 120.0, "completions/min_length": 113.0, "completions/max_length": 135.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 120.0, "completions/min_terminated_length": 113.0, "completions/max_terminated_length": 135.0, "rewards/meter/mean": 0.3736605644226074, "rewards/meter/std": 0.2752265930175781, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9975237250328064, "rewards/repeat_soft/std": 0.001548389787785709, "rewards/judge_quality/mean": 0.809999942779541, "rewards/judge_quality/std": 0.19146056473255157, "rewards/total_composite/mean": 0.5476093888282776, "rewards/total_composite/std": 0.16478323936462402, "reward": 0.5476093888282776, "reward_std": 0.16478323936462402, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11544646322727203, "sampling/sampling_logp_difference/max": 1.9932332038879395, "sampling/importance_sampling_ratio/min": 0.13625416159629822, "sampling/importance_sampling_ratio/mean": 1.0179648399353027, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7649307884275913, "clip_ratio/low_mean": 0.0729277259670198, "clip_ratio/low_min": 0.0729277259670198, "clip_ratio/high_mean": 0.0333252027630806, "clip_ratio/high_max": 0.0333252027630806, "clip_ratio/region_mean": 0.1062529287301004, "reward_total_mean": 0.5476093888282776, "reward_meter_mean": 0.3736605644226074, "reward_meter_std": 0.2752265930175781, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9975237250328064, "reward_repeat_soft_std": 0.001548389787785709, "reward_judge_quality_mean": 0.809999942779541, "reward_judge_quality_std": 0.19146056473255157, "reward_total_composite_mean": 0.5476093888282776, "reward_total_composite_std": 0.16478323936462402} {"timestamp_utc": "2026-04-13T07:24:04Z", "mode": "train", "global_step": 26, "epoch": 0.002611752887995982, "loss": 0.0341, "grad_norm": 11.169930458068848, "learning_rate": 9.924242424242425e-06, "num_tokens": 48753.0, "completions/mean_length": 32.25, "completions/min_length": 31.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.25, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9985294342041016, "rewards/meter/std": 0.0007957193301990628, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.6169735193252563, "rewards/total_composite/std": 0.00021723672398366034, "reward": 0.6169735193252563, "reward_std": 0.0002172399399569258, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.044476937502622604, "sampling/sampling_logp_difference/max": 0.9176238775253296, "sampling/importance_sampling_ratio/min": 0.3994671106338501, "sampling/importance_sampling_ratio/mean": 1.0029505491256714, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2671094238758087, "clip_ratio/low_mean": 0.01785714365541935, "clip_ratio/low_min": 0.01785714365541935, "clip_ratio/high_mean": 0.02726221503689885, "clip_ratio/high_max": 0.02726221503689885, "clip_ratio/region_mean": 0.0451193586923182, "reward_total_mean": 0.6169735193252563, "reward_meter_mean": 0.9985294342041016, "reward_meter_std": 0.0007957193301990628, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.6169735193252563, "reward_total_composite_std": 0.00021723672398366034} {"timestamp_utc": "2026-04-13T07:24:12Z", "mode": "train", "global_step": 27, "epoch": 0.0027122049221496736, "loss": -0.0472, "grad_norm": 8.745513916015625, "learning_rate": 9.921212121212121e-06, "num_tokens": 51435.0, "completions/mean_length": 138.25, "completions/min_length": 83.0, "completions/max_length": 191.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 138.25, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 191.0, "rewards/meter/mean": 0.45913514494895935, "rewards/meter/std": 0.3512060344219208, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.999238133430481, "rewards/repeat_soft/std": 0.0007623151177540421, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.20469054579734802, "rewards/total_composite/mean": 0.43480345606803894, "rewards/total_composite/std": 0.19026042520999908, "reward": 0.43480345606803894, "reward_std": 0.1902604103088379, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2304992526769638, "sampling/sampling_logp_difference/max": 1.7399463653564453, "sampling/importance_sampling_ratio/min": 0.17552980780601501, "sampling/importance_sampling_ratio/mean": 1.04105544090271, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.5277575254440308, "clip_ratio/low_mean": 0.07658866792917252, "clip_ratio/low_min": 0.07658866792917252, "clip_ratio/high_mean": 0.15760960429906845, "clip_ratio/high_max": 0.15760960429906845, "clip_ratio/region_mean": 0.23419827222824097, "reward_total_mean": 0.43480345606803894, "reward_meter_mean": 0.45913514494895935, "reward_meter_std": 0.3512060344219208, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.999238133430481, "reward_repeat_soft_std": 0.0007623151177540421, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.20469054579734802, "reward_total_composite_mean": 0.43480345606803894, "reward_total_composite_std": 0.19026042520999908} {"timestamp_utc": "2026-04-13T07:24:19Z", "mode": "train", "global_step": 28, "epoch": 0.0028126569563033652, "loss": 0.0977, "grad_norm": 23.730607986450195, "learning_rate": 9.918181818181818e-06, "num_tokens": 52963.0, "completions/mean_length": 29.0, "completions/min_length": 16.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.0, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.6324816346168518, "rewards/meter/std": 0.4750705063343048, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9585226774215698, "rewards/repeat_soft/std": 0.011249415576457977, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.5238929390907288, "rewards/total_composite/std": 0.13481268286705017, "reward": 0.5238929390907288, "reward_std": 0.13481266796588898, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19141219556331635, "sampling/sampling_logp_difference/max": 1.2272367477416992, "sampling/importance_sampling_ratio/min": 0.29310137033462524, "sampling/importance_sampling_ratio/mean": 1.0181347131729126, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8611400574445724, "clip_ratio/low_mean": 0.07508680690079927, "clip_ratio/low_min": 0.07508680690079927, "clip_ratio/high_mean": 0.15117423422634602, "clip_ratio/high_max": 0.15117423422634602, "clip_ratio/region_mean": 0.2262610411271453, "reward_total_mean": 0.5238929390907288, "reward_meter_mean": 0.6324816346168518, "reward_meter_std": 0.4750705063343048, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9585226774215698, "reward_repeat_soft_std": 0.011249415576457977, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.5238929390907288, "reward_total_composite_std": 0.13481268286705017} {"timestamp_utc": "2026-04-13T07:24:28Z", "mode": "train", "global_step": 29, "epoch": 0.002913108990457057, "loss": 0.062, "grad_norm": 7.373910427093506, "learning_rate": 9.915151515151515e-06, "num_tokens": 56000.0, "completions/mean_length": 179.625, "completions/min_length": 150.0, "completions/max_length": 217.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 179.625, "completions/min_terminated_length": 150.0, "completions/max_terminated_length": 217.0, "rewards/meter/mean": 0.8581138253211975, "rewards/meter/std": 0.28778037428855896, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9989545345306396, "rewards/repeat_soft/std": 0.000916292832698673, "rewards/judge_quality/mean": 0.41499999165534973, "rewards/judge_quality/std": 0.14520922303199768, "rewards/total_composite/mean": 0.5919029712677002, "rewards/total_composite/std": 0.11980368196964264, "reward": 0.5919029712677002, "reward_std": 0.11980368942022324, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20266062021255493, "sampling/sampling_logp_difference/max": 1.7890448570251465, "sampling/importance_sampling_ratio/min": 0.16711971163749695, "sampling/importance_sampling_ratio/mean": 1.0425152778625488, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3875505477190018, "clip_ratio/low_mean": 0.07503936253488064, "clip_ratio/low_min": 0.07503936253488064, "clip_ratio/high_mean": 0.13527059368789196, "clip_ratio/high_max": 0.13527059368789196, "clip_ratio/region_mean": 0.2103099562227726, "reward_total_mean": 0.5919029712677002, "reward_meter_mean": 0.8581138253211975, "reward_meter_std": 0.28778037428855896, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9989545345306396, "reward_repeat_soft_std": 0.000916292832698673, "reward_judge_quality_mean": 0.41499999165534973, "reward_judge_quality_std": 0.14520922303199768, "reward_total_composite_mean": 0.5919029712677002, "reward_total_composite_std": 0.11980368196964264} {"timestamp_utc": "2026-04-13T07:24:36Z", "mode": "train", "global_step": 30, "epoch": 0.0030135610246107484, "loss": -0.0097, "grad_norm": 12.832189559936523, "learning_rate": 9.912121212121213e-06, "num_tokens": 57955.0, "completions/mean_length": 60.375, "completions/min_length": 37.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.375, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.4923819899559021, "rewards/meter/std": 0.4194702208042145, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.998965859413147, "rewards/repeat_soft/std": 0.002180325798690319, "rewards/judge_quality/mean": 0.5199999809265137, "rewards/judge_quality/std": 0.19272483885288239, "rewards/total_composite/mean": 0.5113567113876343, "rewards/total_composite/std": 0.15652590990066528, "reward": 0.5113567113876343, "reward_std": 0.15652593970298767, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2167627066373825, "sampling/sampling_logp_difference/max": 1.6749815940856934, "sampling/importance_sampling_ratio/min": 0.18731161952018738, "sampling/importance_sampling_ratio/mean": 1.0154328346252441, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.061600148677826, "clip_ratio/low_mean": 0.1053523225709796, "clip_ratio/low_min": 0.1053523225709796, "clip_ratio/high_mean": 0.09013216756284237, "clip_ratio/high_max": 0.09013216756284237, "clip_ratio/region_mean": 0.19548449013382196, "reward_total_mean": 0.5113567113876343, "reward_meter_mean": 0.4923819899559021, "reward_meter_std": 0.4194702208042145, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.998965859413147, "reward_repeat_soft_std": 0.002180325798690319, "reward_judge_quality_mean": 0.5199999809265137, "reward_judge_quality_std": 0.19272483885288239, "reward_total_composite_mean": 0.5113567113876343, "reward_total_composite_std": 0.15652590990066528} {"timestamp_utc": "2026-04-13T07:24:43Z", "mode": "train", "global_step": 31, "epoch": 0.00311401305876444, "loss": 0.0774, "grad_norm": 17.354585647583008, "learning_rate": 9.90909090909091e-06, "num_tokens": 59958.0, "completions/mean_length": 78.375, "completions/min_length": 60.0, "completions/max_length": 97.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.375, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.3928065001964569, "rewards/meter/std": 0.29107925295829773, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9918286800384521, "rewards/repeat_soft/std": 0.006046677939593792, "rewards/judge_quality/mean": 0.5199999809265137, "rewards/judge_quality/std": 0.19272483885288239, "rewards/total_composite/mean": 0.46345409750938416, "rewards/total_composite/std": 0.07556276023387909, "reward": 0.46345409750938416, "reward_std": 0.07556275278329849, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23961949348449707, "sampling/sampling_logp_difference/max": 2.6145894527435303, "sampling/importance_sampling_ratio/min": 0.07319783419370651, "sampling/importance_sampling_ratio/mean": 0.9922016263008118, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2000147253274918, "clip_ratio/low_mean": 0.09721448086202145, "clip_ratio/low_min": 0.09721448086202145, "clip_ratio/high_mean": 0.10816307179629803, "clip_ratio/high_max": 0.10816307179629803, "clip_ratio/region_mean": 0.20537755265831947, "reward_total_mean": 0.46345409750938416, "reward_meter_mean": 0.3928065001964569, "reward_meter_std": 0.29107925295829773, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9918286800384521, "reward_repeat_soft_std": 0.006046677939593792, "reward_judge_quality_mean": 0.5199999809265137, "reward_judge_quality_std": 0.19272483885288239, "reward_total_composite_mean": 0.46345409750938416, "reward_total_composite_std": 0.07556276023387909} {"timestamp_utc": "2026-04-13T07:24:50Z", "mode": "train", "global_step": 32, "epoch": 0.0032144650929181316, "loss": 0.0951, "grad_norm": 33.8635139465332, "learning_rate": 9.906060606060607e-06, "num_tokens": 61400.0, "completions/mean_length": 25.25, "completions/min_length": 17.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.25, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.2765427529811859, "rewards/meter/std": 0.24548844993114471, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.7437499761581421, "rewards/judge_quality/std": 0.2432481348514557, "rewards/total_composite/mean": 0.47309568524360657, "rewards/total_composite/std": 0.12770362198352814, "reward": 0.47309568524360657, "reward_std": 0.12770362198352814, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24985970556735992, "sampling/sampling_logp_difference/max": 1.6838467121124268, "sampling/importance_sampling_ratio/min": 0.18565842509269714, "sampling/importance_sampling_ratio/mean": 0.9918996691703796, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.68735009431839, "clip_ratio/low_mean": 0.08237721398472786, "clip_ratio/low_min": 0.08237721398472786, "clip_ratio/high_mean": 0.10065257363021374, "clip_ratio/high_max": 0.10065257363021374, "clip_ratio/region_mean": 0.1830297876149416, "reward_total_mean": 0.47309568524360657, "reward_meter_mean": 0.2765427529811859, "reward_meter_std": 0.24548844993114471, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.7437499761581421, "reward_judge_quality_std": 0.2432481348514557, "reward_total_composite_mean": 0.47309568524360657, "reward_total_composite_std": 0.12770362198352814} {"timestamp_utc": "2026-04-13T07:24:58Z", "mode": "train", "global_step": 33, "epoch": 0.0033149171270718232, "loss": -0.0659, "grad_norm": 18.40517234802246, "learning_rate": 9.903030303030305e-06, "num_tokens": 63006.0, "completions/mean_length": 36.75, "completions/min_length": 25.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.75, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.47970691323280334, "rewards/meter/std": 0.39111989736557007, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9984102249145508, "rewards/repeat_soft/std": 0.0024311586748808622, "rewards/judge_quality/mean": 0.5950000286102295, "rewards/judge_quality/std": 0.19820626080036163, "rewards/total_composite/mean": 0.45416176319122314, "rewards/total_composite/std": 0.22378690540790558, "reward": 0.45416176319122314, "reward_std": 0.22378690540790558, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20951895415782928, "sampling/sampling_logp_difference/max": 1.2337064743041992, "sampling/importance_sampling_ratio/min": 0.29121121764183044, "sampling/importance_sampling_ratio/mean": 1.0673010349273682, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.090721681714058, "clip_ratio/low_mean": 0.07213383819907904, "clip_ratio/low_min": 0.07213383819907904, "clip_ratio/high_mean": 0.12102419137954712, "clip_ratio/high_max": 0.12102419137954712, "clip_ratio/region_mean": 0.19315802957862616, "reward_total_mean": 0.45416176319122314, "reward_meter_mean": 0.47970691323280334, "reward_meter_std": 0.39111989736557007, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9984102249145508, "reward_repeat_soft_std": 0.0024311586748808622, "reward_judge_quality_mean": 0.5950000286102295, "reward_judge_quality_std": 0.19820626080036163, "reward_total_composite_mean": 0.45416176319122314, "reward_total_composite_std": 0.22378690540790558} {"timestamp_utc": "2026-04-13T07:25:05Z", "mode": "train", "global_step": 34, "epoch": 0.003415369161225515, "loss": 0.0987, "grad_norm": 16.91385841369629, "learning_rate": 9.9e-06, "num_tokens": 64599.0, "completions/mean_length": 41.125, "completions/min_length": 27.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.125, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.3966813385486603, "rewards/meter/std": 0.3952701985836029, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9912532567977905, "rewards/repeat_soft/std": 0.009152887389063835, "rewards/judge_quality/mean": 0.690000057220459, "rewards/judge_quality/std": 0.22315914928913116, "rewards/total_composite/mean": 0.5417426824569702, "rewards/total_composite/std": 0.21540047228336334, "reward": 0.5417426824569702, "reward_std": 0.21540047228336334, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2257329225540161, "sampling/sampling_logp_difference/max": 2.2196807861328125, "sampling/importance_sampling_ratio/min": 0.1086437851190567, "sampling/importance_sampling_ratio/mean": 1.0293428897857666, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3586771450936794, "clip_ratio/low_mean": 0.1340062189847231, "clip_ratio/low_min": 0.1340062189847231, "clip_ratio/high_mean": 0.07132523134350777, "clip_ratio/high_max": 0.07132523134350777, "clip_ratio/region_mean": 0.20533145032823086, "reward_total_mean": 0.5417426824569702, "reward_meter_mean": 0.3966813385486603, "reward_meter_std": 0.3952701985836029, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9912532567977905, "reward_repeat_soft_std": 0.009152887389063835, "reward_judge_quality_mean": 0.690000057220459, "reward_judge_quality_std": 0.22315914928913116, "reward_total_composite_mean": 0.5417426824569702, "reward_total_composite_std": 0.21540047228336334} {"timestamp_utc": "2026-04-13T07:25:11Z", "mode": "train", "global_step": 35, "epoch": 0.0035158211953792064, "loss": 0.0704, "grad_norm": 27.953187942504883, "learning_rate": 9.896969696969699e-06, "num_tokens": 66168.0, "completions/mean_length": 31.125, "completions/min_length": 24.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.125, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.7837445735931396, "rewards/meter/std": 0.2611982822418213, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9804645776748657, "rewards/repeat_soft/std": 0.03129582852125168, "rewards/judge_quality/mean": 0.6100000143051147, "rewards/judge_quality/std": 0.15250293910503387, "rewards/total_composite/mean": 0.6702450513839722, "rewards/total_composite/std": 0.1414785236120224, "reward": 0.6702450513839722, "reward_std": 0.1414785236120224, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1811247318983078, "sampling/sampling_logp_difference/max": 1.859018325805664, "sampling/importance_sampling_ratio/min": 0.15582552552223206, "sampling/importance_sampling_ratio/mean": 0.959926187992096, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6075849942862988, "clip_ratio/low_mean": 0.06071342155337334, "clip_ratio/low_min": 0.06071342155337334, "clip_ratio/high_mean": 0.083756223320961, "clip_ratio/high_max": 0.083756223320961, "clip_ratio/region_mean": 0.14446964487433434, "reward_total_mean": 0.6702450513839722, "reward_meter_mean": 0.7837445735931396, "reward_meter_std": 0.2611982822418213, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9804645776748657, "reward_repeat_soft_std": 0.03129582852125168, "reward_judge_quality_mean": 0.6100000143051147, "reward_judge_quality_std": 0.15250293910503387, "reward_total_composite_mean": 0.6702450513839722, "reward_total_composite_std": 0.1414785236120224} {"timestamp_utc": "2026-04-13T07:25:18Z", "mode": "train", "global_step": 36, "epoch": 0.003616273229532898, "loss": 0.0955, "grad_norm": 15.208176612854004, "learning_rate": 9.893939393939395e-06, "num_tokens": 67682.0, "completions/mean_length": 40.25, "completions/min_length": 32.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.25, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.4485693871974945, "rewards/meter/std": 0.3291569948196411, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9953606724739075, "rewards/repeat_soft/std": 0.0055753388442099094, "rewards/judge_quality/mean": 0.7350000143051147, "rewards/judge_quality/std": 0.15014280378818512, "rewards/total_composite/mean": 0.5608401298522949, "rewards/total_composite/std": 0.15952417254447937, "reward": 0.5608401298522949, "reward_std": 0.15952417254447937, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18473203480243683, "sampling/sampling_logp_difference/max": 1.8001375198364258, "sampling/importance_sampling_ratio/min": 0.1652761697769165, "sampling/importance_sampling_ratio/mean": 1.016972541809082, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4089657962322235, "clip_ratio/low_mean": 0.07581416238099337, "clip_ratio/low_min": 0.07581416238099337, "clip_ratio/high_mean": 0.08979000151157379, "clip_ratio/high_max": 0.08979000151157379, "clip_ratio/region_mean": 0.16560416389256716, "reward_total_mean": 0.5608401298522949, "reward_meter_mean": 0.4485693871974945, "reward_meter_std": 0.3291569948196411, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9953606724739075, "reward_repeat_soft_std": 0.0055753388442099094, "reward_judge_quality_mean": 0.7350000143051147, "reward_judge_quality_std": 0.15014280378818512, "reward_total_composite_mean": 0.5608401298522949, "reward_total_composite_std": 0.15952417254447937} {"timestamp_utc": "2026-04-13T07:25:24Z", "mode": "train", "global_step": 37, "epoch": 0.0037167252636865896, "loss": -0.0345, "grad_norm": 15.821805000305176, "learning_rate": 9.890909090909092e-06, "num_tokens": 69454.0, "completions/mean_length": 50.5, "completions/min_length": 34.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.5, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.7323452234268188, "rewards/meter/std": 0.3667115867137909, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9898346066474915, "rewards/repeat_soft/std": 0.013633593916893005, "rewards/judge_quality/mean": 0.5487500429153442, "rewards/judge_quality/std": 0.22937415540218353, "rewards/total_composite/mean": 0.6005253195762634, "rewards/total_composite/std": 0.16267496347427368, "reward": 0.6005253195762634, "reward_std": 0.16267496347427368, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22236742079257965, "sampling/sampling_logp_difference/max": 1.5800833702087402, "sampling/importance_sampling_ratio/min": 0.20595793426036835, "sampling/importance_sampling_ratio/mean": 1.0082014799118042, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.157983884215355, "clip_ratio/low_mean": 0.10209287982434034, "clip_ratio/low_min": 0.10209287982434034, "clip_ratio/high_mean": 0.11077776551246643, "clip_ratio/high_max": 0.11077776551246643, "clip_ratio/region_mean": 0.21287064533680677, "reward_total_mean": 0.6005253195762634, "reward_meter_mean": 0.7323452234268188, "reward_meter_std": 0.3667115867137909, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9898346066474915, "reward_repeat_soft_std": 0.013633593916893005, "reward_judge_quality_mean": 0.5487500429153442, "reward_judge_quality_std": 0.22937415540218353, "reward_total_composite_mean": 0.6005253195762634, "reward_total_composite_std": 0.16267496347427368} {"timestamp_utc": "2026-04-13T07:25:31Z", "mode": "train", "global_step": 38, "epoch": 0.0038171772978402812, "loss": 0.0838, "grad_norm": 16.704303741455078, "learning_rate": 9.887878787878789e-06, "num_tokens": 71106.0, "completions/mean_length": 53.5, "completions/min_length": 33.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.5, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.5510157942771912, "rewards/meter/std": 0.41519367694854736, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9980113506317139, "rewards/repeat_soft/std": 0.005624707788228989, "rewards/judge_quality/mean": 0.5099999904632568, "rewards/judge_quality/std": 0.13979578018188477, "rewards/total_composite/mean": 0.5337550640106201, "rewards/total_composite/std": 0.1366468220949173, "reward": 0.5337550640106201, "reward_std": 0.1366468071937561, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19399528205394745, "sampling/sampling_logp_difference/max": 1.4106621742248535, "sampling/importance_sampling_ratio/min": 0.24398167431354523, "sampling/importance_sampling_ratio/mean": 1.0123108625411987, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4564415216445923, "clip_ratio/low_mean": 0.07825370226055384, "clip_ratio/low_min": 0.07825370226055384, "clip_ratio/high_mean": 0.11193818785250187, "clip_ratio/high_max": 0.11193818785250187, "clip_ratio/region_mean": 0.1901918901130557, "reward_total_mean": 0.5337550640106201, "reward_meter_mean": 0.5510157942771912, "reward_meter_std": 0.41519367694854736, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9980113506317139, "reward_repeat_soft_std": 0.005624707788228989, "reward_judge_quality_mean": 0.5099999904632568, "reward_judge_quality_std": 0.13979578018188477, "reward_total_composite_mean": 0.5337550640106201, "reward_total_composite_std": 0.1366468220949173} {"timestamp_utc": "2026-04-13T07:25:39Z", "mode": "train", "global_step": 39, "epoch": 0.003917629331993973, "loss": 0.1019, "grad_norm": 35.369117736816406, "learning_rate": 9.884848484848486e-06, "num_tokens": 72810.0, "completions/mean_length": 55.0, "completions/min_length": 48.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.0, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.770334005355835, "rewards/meter/std": 0.22616279125213623, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9931648969650269, "rewards/repeat_soft/std": 0.008299742825329304, "rewards/judge_quality/mean": 0.6524999737739563, "rewards/judge_quality/std": 0.20190168917179108, "rewards/total_composite/mean": 0.685139536857605, "rewards/total_composite/std": 0.15254512429237366, "reward": 0.685139536857605, "reward_std": 0.15254512429237366, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14268994331359863, "sampling/sampling_logp_difference/max": 2.045924663543701, "sampling/importance_sampling_ratio/min": 0.1292606145143509, "sampling/importance_sampling_ratio/mean": 1.000756025314331, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5295688956975937, "clip_ratio/low_mean": 0.06531935092061758, "clip_ratio/low_min": 0.06531935092061758, "clip_ratio/high_mean": 0.04677868401631713, "clip_ratio/high_max": 0.04677868401631713, "clip_ratio/region_mean": 0.11209803493693471, "reward_total_mean": 0.685139536857605, "reward_meter_mean": 0.770334005355835, "reward_meter_std": 0.22616279125213623, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9931648969650269, "reward_repeat_soft_std": 0.008299742825329304, "reward_judge_quality_mean": 0.6524999737739563, "reward_judge_quality_std": 0.20190168917179108, "reward_total_composite_mean": 0.685139536857605, "reward_total_composite_std": 0.15254512429237366} {"timestamp_utc": "2026-04-13T07:25:46Z", "mode": "train", "global_step": 40, "epoch": 0.004018081366147665, "loss": 0.2088, "grad_norm": 23.99601173400879, "learning_rate": 9.881818181818182e-06, "num_tokens": 74420.0, "completions/mean_length": 38.25, "completions/min_length": 28.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.25, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.6893784403800964, "rewards/meter/std": 0.4171596169471741, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9735429286956787, "rewards/repeat_soft/std": 0.04065443575382233, "rewards/judge_quality/mean": 0.8825000524520874, "rewards/judge_quality/std": 0.07440238445997238, "rewards/total_composite/mean": 0.757306694984436, "rewards/total_composite/std": 0.2491140067577362, "reward": 0.757306694984436, "reward_std": 0.24911397695541382, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2222781926393509, "sampling/sampling_logp_difference/max": 2.1597957611083984, "sampling/importance_sampling_ratio/min": 0.115348681807518, "sampling/importance_sampling_ratio/mean": 1.0162009000778198, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0633226111531258, "clip_ratio/low_mean": 0.08311870507895947, "clip_ratio/low_min": 0.08311870507895947, "clip_ratio/high_mean": 0.09333791490644217, "clip_ratio/high_max": 0.09333791490644217, "clip_ratio/region_mean": 0.17645661998540163, "reward_total_mean": 0.757306694984436, "reward_meter_mean": 0.6893784403800964, "reward_meter_std": 0.4171596169471741, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9735429286956787, "reward_repeat_soft_std": 0.04065443575382233, "reward_judge_quality_mean": 0.8825000524520874, "reward_judge_quality_std": 0.07440238445997238, "reward_total_composite_mean": 0.757306694984436, "reward_total_composite_std": 0.2491140067577362} {"timestamp_utc": "2026-04-13T07:25:54Z", "mode": "train", "global_step": 41, "epoch": 0.004118533400301356, "loss": 0.1591, "grad_norm": 14.506078720092773, "learning_rate": 9.87878787878788e-06, "num_tokens": 76186.0, "completions/mean_length": 60.75, "completions/min_length": 33.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.75, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.6156733632087708, "rewards/meter/std": 0.4310094714164734, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9923155903816223, "rewards/repeat_soft/std": 0.013626428321003914, "rewards/judge_quality/mean": 0.45124998688697815, "rewards/judge_quality/std": 0.12799973785877228, "rewards/total_composite/mean": 0.5048567056655884, "rewards/total_composite/std": 0.14636343717575073, "reward": 0.5048567056655884, "reward_std": 0.14636343717575073, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20026902854442596, "sampling/sampling_logp_difference/max": 1.480910301208496, "sampling/importance_sampling_ratio/min": 0.2274305820465088, "sampling/importance_sampling_ratio/mean": 1.0431978702545166, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1705337911844254, "clip_ratio/low_mean": 0.06033051759004593, "clip_ratio/low_min": 0.06033051759004593, "clip_ratio/high_mean": 0.10448822379112244, "clip_ratio/high_max": 0.10448822379112244, "clip_ratio/region_mean": 0.16481874138116837, "reward_total_mean": 0.5048567056655884, "reward_meter_mean": 0.6156733632087708, "reward_meter_std": 0.4310094714164734, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9923155903816223, "reward_repeat_soft_std": 0.013626428321003914, "reward_judge_quality_mean": 0.45124998688697815, "reward_judge_quality_std": 0.12799973785877228, "reward_total_composite_mean": 0.5048567056655884, "reward_total_composite_std": 0.14636343717575073} {"timestamp_utc": "2026-04-13T07:26:01Z", "mode": "train", "global_step": 42, "epoch": 0.004218985434455048, "loss": -0.0256, "grad_norm": 16.84304428100586, "learning_rate": 9.875757575757576e-06, "num_tokens": 77725.0, "completions/mean_length": 43.375, "completions/min_length": 25.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.375, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.582551121711731, "rewards/meter/std": 0.4320070743560791, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.977229654788971, "rewards/repeat_soft/std": 0.020586658269166946, "rewards/judge_quality/mean": 0.5649999976158142, "rewards/judge_quality/std": 0.3022770881652832, "rewards/total_composite/mean": 0.5598830580711365, "rewards/total_composite/std": 0.3062882721424103, "reward": 0.5598830580711365, "reward_std": 0.3062882423400879, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21116963028907776, "sampling/sampling_logp_difference/max": 2.439089059829712, "sampling/importance_sampling_ratio/min": 0.08724027872085571, "sampling/importance_sampling_ratio/mean": 1.0060738325119019, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6929726600646973, "clip_ratio/low_mean": 0.0896103922277689, "clip_ratio/low_min": 0.0896103922277689, "clip_ratio/high_mean": 0.16354099847376347, "clip_ratio/high_max": 0.16354099847376347, "clip_ratio/region_mean": 0.25315139070153236, "reward_total_mean": 0.5598830580711365, "reward_meter_mean": 0.582551121711731, "reward_meter_std": 0.4320070743560791, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.977229654788971, "reward_repeat_soft_std": 0.020586658269166946, "reward_judge_quality_mean": 0.5649999976158142, "reward_judge_quality_std": 0.3022770881652832, "reward_total_composite_mean": 0.5598830580711365, "reward_total_composite_std": 0.3062882721424103} {"timestamp_utc": "2026-04-13T07:26:08Z", "mode": "train", "global_step": 43, "epoch": 0.004319437468608739, "loss": -0.0522, "grad_norm": 17.400814056396484, "learning_rate": 9.872727272727274e-06, "num_tokens": 79484.0, "completions/mean_length": 44.875, "completions/min_length": 31.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.875, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.624340295791626, "rewards/meter/std": 0.4104929268360138, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.997374415397644, "rewards/repeat_soft/std": 0.004505294840782881, "rewards/judge_quality/mean": 0.7224999666213989, "rewards/judge_quality/std": 0.232854962348938, "rewards/total_composite/mean": 0.6401236653327942, "rewards/total_composite/std": 0.20709064602851868, "reward": 0.6401236653327942, "reward_std": 0.2070906162261963, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22349603474140167, "sampling/sampling_logp_difference/max": 1.757357120513916, "sampling/importance_sampling_ratio/min": 0.17250016331672668, "sampling/importance_sampling_ratio/mean": 1.0200906991958618, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7742891013622284, "clip_ratio/low_mean": 0.12815146706998348, "clip_ratio/low_min": 0.12815146706998348, "clip_ratio/high_mean": 0.08952812105417252, "clip_ratio/high_max": 0.08952812105417252, "clip_ratio/region_mean": 0.217679588124156, "reward_total_mean": 0.6401236653327942, "reward_meter_mean": 0.624340295791626, "reward_meter_std": 0.4104929268360138, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.997374415397644, "reward_repeat_soft_std": 0.004505294840782881, "reward_judge_quality_mean": 0.7224999666213989, "reward_judge_quality_std": 0.232854962348938, "reward_total_composite_mean": 0.6401236653327942, "reward_total_composite_std": 0.20709064602851868} {"timestamp_utc": "2026-04-13T07:26:16Z", "mode": "train", "global_step": 44, "epoch": 0.004419889502762431, "loss": -0.0165, "grad_norm": 11.687650680541992, "learning_rate": 9.869696969696971e-06, "num_tokens": 81723.0, "completions/mean_length": 99.875, "completions/min_length": 74.0, "completions/max_length": 128.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 99.875, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 128.0, "rewards/meter/mean": 0.3675351142883301, "rewards/meter/std": 0.2946922183036804, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9973076581954956, "rewards/repeat_soft/std": 0.002466563368216157, "rewards/judge_quality/mean": 0.5325000286102295, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.3810186982154846, "rewards/total_composite/std": 0.1699932962656021, "reward": 0.3810186982154846, "reward_std": 0.16999328136444092, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20723651349544525, "sampling/sampling_logp_difference/max": 1.8403129577636719, "sampling/importance_sampling_ratio/min": 0.1587677299976349, "sampling/importance_sampling_ratio/mean": 1.039668083190918, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.169490233063698, "clip_ratio/low_mean": 0.06495124101638794, "clip_ratio/low_min": 0.06495124101638794, "clip_ratio/high_mean": 0.11929397657513618, "clip_ratio/high_max": 0.11929397657513618, "clip_ratio/region_mean": 0.18424521759152412, "reward_total_mean": 0.3810186982154846, "reward_meter_mean": 0.3675351142883301, "reward_meter_std": 0.2946922183036804, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9973076581954956, "reward_repeat_soft_std": 0.002466563368216157, "reward_judge_quality_mean": 0.5325000286102295, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.3810186982154846, "reward_total_composite_std": 0.1699932962656021} {"timestamp_utc": "2026-04-13T07:26:23Z", "mode": "train", "global_step": 45, "epoch": 0.0045203415369161224, "loss": 0.1585, "grad_norm": 14.383469581604004, "learning_rate": 9.866666666666668e-06, "num_tokens": 83331.0, "completions/mean_length": 50.0, "completions/min_length": 38.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.0, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.7773149013519287, "rewards/meter/std": 0.25354528427124023, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9865143895149231, "rewards/repeat_soft/std": 0.02192966639995575, "rewards/judge_quality/mean": 0.6150000095367432, "rewards/judge_quality/std": 0.25286927819252014, "rewards/total_composite/mean": 0.5853496193885803, "rewards/total_composite/std": 0.2865021228790283, "reward": 0.5853496193885803, "reward_std": 0.2865021228790283, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20703330636024475, "sampling/sampling_logp_difference/max": 1.642387866973877, "sampling/importance_sampling_ratio/min": 0.19351740181446075, "sampling/importance_sampling_ratio/mean": 1.0344429016113281, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8182155936956406, "clip_ratio/low_mean": 0.08375179395079613, "clip_ratio/low_min": 0.08375179395079613, "clip_ratio/high_mean": 0.12188977561891079, "clip_ratio/high_max": 0.12188977561891079, "clip_ratio/region_mean": 0.20564156956970692, "reward_total_mean": 0.5853496193885803, "reward_meter_mean": 0.7773149013519287, "reward_meter_std": 0.25354528427124023, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9865143895149231, "reward_repeat_soft_std": 0.02192966639995575, "reward_judge_quality_mean": 0.6150000095367432, "reward_judge_quality_std": 0.25286927819252014, "reward_total_composite_mean": 0.5853496193885803, "reward_total_composite_std": 0.2865021228790283} {"timestamp_utc": "2026-04-13T07:26:31Z", "mode": "train", "global_step": 46, "epoch": 0.0046207935710698145, "loss": -0.0718, "grad_norm": 15.25479793548584, "learning_rate": 9.863636363636364e-06, "num_tokens": 84939.0, "completions/mean_length": 45.0, "completions/min_length": 36.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.0, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.5657721757888794, "rewards/meter/std": 0.4256070852279663, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9984498023986816, "rewards/repeat_soft/std": 0.004384705331176519, "rewards/judge_quality/mean": 0.6775000095367432, "rewards/judge_quality/std": 0.25949129462242126, "rewards/total_composite/mean": 0.6238999366760254, "rewards/total_composite/std": 0.23371422290802002, "reward": 0.6238999366760254, "reward_std": 0.23371422290802002, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2331676185131073, "sampling/sampling_logp_difference/max": 1.3806533813476562, "sampling/importance_sampling_ratio/min": 0.2514142394065857, "sampling/importance_sampling_ratio/mean": 1.0120073556900024, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.181841418147087, "clip_ratio/low_mean": 0.08811917342245579, "clip_ratio/low_min": 0.08811917342245579, "clip_ratio/high_mean": 0.13466385751962662, "clip_ratio/high_max": 0.13466385751962662, "clip_ratio/region_mean": 0.2227830309420824, "reward_total_mean": 0.6238999366760254, "reward_meter_mean": 0.5657721757888794, "reward_meter_std": 0.4256070852279663, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9984498023986816, "reward_repeat_soft_std": 0.004384705331176519, "reward_judge_quality_mean": 0.6775000095367432, "reward_judge_quality_std": 0.25949129462242126, "reward_total_composite_mean": 0.6238999366760254, "reward_total_composite_std": 0.23371422290802002} {"timestamp_utc": "2026-04-13T07:26:37Z", "mode": "train", "global_step": 47, "epoch": 0.004721245605223506, "loss": 0.0598, "grad_norm": 19.374162673950195, "learning_rate": 9.860606060606061e-06, "num_tokens": 86499.0, "completions/mean_length": 46.0, "completions/min_length": 35.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.26104703545570374, "rewards/meter/std": 0.2333628535270691, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9847595691680908, "rewards/repeat_soft/std": 0.011016963981091976, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.44352251291275024, "rewards/total_composite/std": 0.11558562517166138, "reward": 0.44352251291275024, "reward_std": 0.11558562517166138, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19819296896457672, "sampling/sampling_logp_difference/max": 2.590005874633789, "sampling/importance_sampling_ratio/min": 0.07501959800720215, "sampling/importance_sampling_ratio/mean": 1.022105097770691, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6669579893350601, "clip_ratio/low_mean": 0.1387309804558754, "clip_ratio/low_min": 0.1387309804558754, "clip_ratio/high_mean": 0.04285714402794838, "clip_ratio/high_max": 0.04285714402794838, "clip_ratio/region_mean": 0.18158812448382378, "reward_total_mean": 0.44352251291275024, "reward_meter_mean": 0.26104703545570374, "reward_meter_std": 0.2333628535270691, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9847595691680908, "reward_repeat_soft_std": 0.011016963981091976, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.44352251291275024, "reward_total_composite_std": 0.11558562517166138} {"timestamp_utc": "2026-04-13T07:26:44Z", "mode": "train", "global_step": 48, "epoch": 0.004821697639377198, "loss": 0.1314, "grad_norm": 28.504535675048828, "learning_rate": 9.857575757575758e-06, "num_tokens": 88108.0, "completions/mean_length": 32.125, "completions/min_length": 28.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.125, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.27756398916244507, "rewards/meter/std": 0.3191877007484436, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9716074466705322, "rewards/repeat_soft/std": 0.036921996623277664, "rewards/judge_quality/mean": 0.6525000333786011, "rewards/judge_quality/std": 0.2921227812767029, "rewards/total_composite/mean": 0.4460914433002472, "rewards/total_composite/std": 0.2519592344760895, "reward": 0.4460914433002472, "reward_std": 0.2519592046737671, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.27054738998413086, "sampling/sampling_logp_difference/max": 4.1249494552612305, "sampling/importance_sampling_ratio/min": 0.01616431027650833, "sampling/importance_sampling_ratio/mean": 1.010880947113037, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2761306390166283, "clip_ratio/low_mean": 0.1557934768497944, "clip_ratio/low_min": 0.1557934768497944, "clip_ratio/high_mean": 0.07254464458674192, "clip_ratio/high_max": 0.07254464458674192, "clip_ratio/region_mean": 0.2283381214365363, "reward_total_mean": 0.4460914433002472, "reward_meter_mean": 0.27756398916244507, "reward_meter_std": 0.3191877007484436, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9716074466705322, "reward_repeat_soft_std": 0.036921996623277664, "reward_judge_quality_mean": 0.6525000333786011, "reward_judge_quality_std": 0.2921227812767029, "reward_total_composite_mean": 0.4460914433002472, "reward_total_composite_std": 0.2519592344760895} {"timestamp_utc": "2026-04-13T07:26:52Z", "mode": "train", "global_step": 49, "epoch": 0.004922149673530889, "loss": 0.0539, "grad_norm": 19.42587661743164, "learning_rate": 9.854545454545456e-06, "num_tokens": 90563.0, "completions/mean_length": 115.875, "completions/min_length": 95.0, "completions/max_length": 136.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.875, "completions/min_terminated_length": 95.0, "completions/max_terminated_length": 136.0, "rewards/meter/mean": 0.30381885170936584, "rewards/meter/std": 0.18320336937904358, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9918557405471802, "rewards/repeat_soft/std": 0.0065759336575865746, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.3872862756252289, "rewards/total_composite/std": 0.16922295093536377, "reward": 0.3872862756252289, "reward_std": 0.16922295093536377, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2825930118560791, "sampling/sampling_logp_difference/max": 3.059009552001953, "sampling/importance_sampling_ratio/min": 0.046934157609939575, "sampling/importance_sampling_ratio/mean": 0.9706795811653137, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2838484272360802, "clip_ratio/low_mean": 0.08543148636817932, "clip_ratio/low_min": 0.08543148636817932, "clip_ratio/high_mean": 0.15653525479137897, "clip_ratio/high_max": 0.15653525479137897, "clip_ratio/region_mean": 0.2419667411595583, "reward_total_mean": 0.3872862756252289, "reward_meter_mean": 0.30381885170936584, "reward_meter_std": 0.18320336937904358, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9918557405471802, "reward_repeat_soft_std": 0.0065759336575865746, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.3872862756252289, "reward_total_composite_std": 0.16922295093536377} {"timestamp_utc": "2026-04-13T07:27:01Z", "mode": "train", "global_step": 50, "epoch": 0.005022601707684581, "loss": -0.0257, "grad_norm": 13.771021842956543, "learning_rate": 9.851515151515151e-06, "num_tokens": 92565.0, "completions/mean_length": 88.25, "completions/min_length": 61.0, "completions/max_length": 123.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 88.25, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.329473614692688, "rewards/meter/std": 0.25273919105529785, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9935731887817383, "rewards/repeat_soft/std": 0.0037070184480398893, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.3710108995437622, "rewards/total_composite/std": 0.15944214165210724, "reward": 0.3710108995437622, "reward_std": 0.15944214165210724, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2445198893547058, "sampling/sampling_logp_difference/max": 2.486546516418457, "sampling/importance_sampling_ratio/min": 0.08319678902626038, "sampling/importance_sampling_ratio/mean": 1.0304960012435913, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.181565210223198, "clip_ratio/low_mean": 0.08294316753745079, "clip_ratio/low_min": 0.08294316753745079, "clip_ratio/high_mean": 0.164827523753047, "clip_ratio/high_max": 0.164827523753047, "clip_ratio/region_mean": 0.24777069129049778, "reward_total_mean": 0.3710108995437622, "reward_meter_mean": 0.329473614692688, "reward_meter_std": 0.25273919105529785, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9935731887817383, "reward_repeat_soft_std": 0.0037070184480398893, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.3710108995437622, "reward_total_composite_std": 0.15944214165210724} {"timestamp_utc": "2026-04-13T07:28:10Z", "mode": "eval", "global_step": 50, "epoch": 0.005022601707684581, "eval_loss": NaN, "eval_runtime": 69.2141, "eval_samples_per_second": 1.156, "eval_steps_per_second": 0.144, "eval_num_tokens": 92565.0, "eval_completions/mean_length": 106.85, "eval_completions/min_length": 31.3, "eval_completions/max_length": 286.3, "eval_completions/clipped_ratio": 0.025, "eval_completions/mean_terminated_length": 96.1928581237793, "eval_completions/min_terminated_length": 31.3, "eval_completions/max_terminated_length": 221.4, "eval_rewards/meter/mean": 0.4507799297571182, "eval_rewards/meter/std": 0.3792104184627533, "eval_rewards/count_adherence/mean": 0.9110416650772095, "eval_rewards/count_adherence/std": 0.20400682613253593, "eval_rewards/hard_gate/mean": 0.95, "eval_rewards/hard_gate/std": 0.11700168251991272, "eval_rewards/repeat_soft/mean": 0.9902224004268646, "eval_rewards/repeat_soft/std": 0.017161866067908704, "eval_rewards/judge_quality/mean": 0.5031250029802322, "eval_rewards/judge_quality/std": 0.18279580399394035, "eval_rewards/total_composite/mean": 0.4718601256608963, "eval_rewards/total_composite/std": 0.17194418162107467, "eval_reward": 0.4718601256608963, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.13539711087942125, "eval_sampling/sampling_logp_difference/max": 1.1591978073120117, "eval_sampling/importance_sampling_ratio/min": 0.31711107939481736, "eval_sampling/importance_sampling_ratio/mean": 1.0372873306274415, "eval_sampling/importance_sampling_ratio/max": 1.536458969116211, "eval_entropy": 2.0635273694992065, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.4718601256608963, "eval_reward_meter_mean": 0.4507799297571182, "eval_reward_meter_std": 0.3792104184627533, "eval_reward_count_adherence_mean": 0.9110416650772095, "eval_reward_count_adherence_std": 0.20400682613253593, "eval_reward_hard_gate_mean": 0.95, "eval_reward_hard_gate_std": 0.11700168251991272, "eval_reward_repeat_soft_mean": 0.9902224004268646, "eval_reward_repeat_soft_std": 0.017161866067908704, "eval_reward_judge_quality_mean": 0.5031250029802322, "eval_reward_judge_quality_std": 0.18279580399394035, "eval_reward_total_composite_mean": 0.4718601256608963, "eval_reward_total_composite_std": 0.17194418162107467} {"timestamp_utc": "2026-04-13T07:28:20Z", "mode": "train", "global_step": 51, "epoch": 0.005123053741838272, "loss": 0.1885, "grad_norm": 15.710267066955566, "learning_rate": 9.84848484848485e-06, "num_tokens": 94303.0, "completions/mean_length": 50.25, "completions/min_length": 33.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.25, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.48370689153671265, "rewards/meter/std": 0.4339752495288849, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9988887906074524, "rewards/repeat_soft/std": 0.002272413345053792, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.5073872208595276, "rewards/total_composite/std": 0.21636413037776947, "reward": 0.5073872208595276, "reward_std": 0.21636416018009186, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2187386006116867, "sampling/sampling_logp_difference/max": 1.439584732055664, "sampling/importance_sampling_ratio/min": 0.2370261698961258, "sampling/importance_sampling_ratio/mean": 1.015249252319336, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4316212087869644, "clip_ratio/low_mean": 0.11987635493278503, "clip_ratio/low_min": 0.11987635493278503, "clip_ratio/high_mean": 0.11116188950836658, "clip_ratio/high_max": 0.11116188950836658, "clip_ratio/region_mean": 0.23103824444115162, "reward_total_mean": 0.5073872208595276, "reward_meter_mean": 0.48370689153671265, "reward_meter_std": 0.4339752495288849, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9988887906074524, "reward_repeat_soft_std": 0.002272413345053792, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.5073872208595276, "reward_total_composite_std": 0.21636413037776947} {"timestamp_utc": "2026-04-13T07:28:33Z", "mode": "train", "global_step": 52, "epoch": 0.005223505775991964, "loss": -0.0651, "grad_norm": 7.459178924560547, "learning_rate": 9.845454545454546e-06, "num_tokens": 96331.0, "completions/mean_length": 143.5, "completions/min_length": 67.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 90.85714721679688, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.5149980783462524, "rewards/meter/std": 0.3162343502044678, "rewards/count_adherence/mean": 0.84375, "rewards/count_adherence/std": 0.35197150707244873, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9875726699829102, "rewards/repeat_soft/std": 0.01723732426762581, "rewards/judge_quality/mean": 0.5487499833106995, "rewards/judge_quality/std": 0.23793382942676544, "rewards/total_composite/mean": 0.43574202060699463, "rewards/total_composite/std": 0.3080213665962219, "reward": 0.43574202060699463, "reward_std": 0.3080213963985443, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2421015501022339, "sampling/sampling_logp_difference/max": 1.4144783020019531, "sampling/importance_sampling_ratio/min": 0.2430523782968521, "sampling/importance_sampling_ratio/mean": 1.0390788316726685, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0853719115257263, "clip_ratio/low_mean": 0.041592495515942574, "clip_ratio/low_min": 0.041592495515942574, "clip_ratio/high_mean": 0.16704162023961544, "clip_ratio/high_max": 0.16704162023961544, "clip_ratio/region_mean": 0.20863411575555801, "reward_total_mean": 0.43574202060699463, "reward_meter_mean": 0.5149980783462524, "reward_meter_std": 0.3162343502044678, "reward_count_adherence_mean": 0.84375, "reward_count_adherence_std": 0.35197150707244873, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9875726699829102, "reward_repeat_soft_std": 0.01723732426762581, "reward_judge_quality_mean": 0.5487499833106995, "reward_judge_quality_std": 0.23793382942676544, "reward_total_composite_mean": 0.43574202060699463, "reward_total_composite_std": 0.3080213665962219} {"timestamp_utc": "2026-04-13T07:28:40Z", "mode": "train", "global_step": 53, "epoch": 0.005323957810145655, "loss": 0.0639, "grad_norm": 29.835546493530273, "learning_rate": 9.842424242424243e-06, "num_tokens": 97818.0, "completions/mean_length": 32.875, "completions/min_length": 25.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.875, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.5110926032066345, "rewards/meter/std": 0.36550405621528625, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9988951683044434, "rewards/repeat_soft/std": 0.0031248305458575487, "rewards/judge_quality/mean": 0.7950000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.5656921863555908, "rewards/total_composite/std": 0.3202142119407654, "reward": 0.5656921863555908, "reward_std": 0.3202142119407654, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20222999155521393, "sampling/sampling_logp_difference/max": 2.314786434173584, "sampling/importance_sampling_ratio/min": 0.09878727793693542, "sampling/importance_sampling_ratio/mean": 0.9813011884689331, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6355803459882736, "clip_ratio/low_mean": 0.10113392397761345, "clip_ratio/low_min": 0.10113392397761345, "clip_ratio/high_mean": 0.051487069576978683, "clip_ratio/high_max": 0.051487069576978683, "clip_ratio/region_mean": 0.15262099355459213, "reward_total_mean": 0.5656921863555908, "reward_meter_mean": 0.5110926032066345, "reward_meter_std": 0.36550405621528625, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9988951683044434, "reward_repeat_soft_std": 0.0031248305458575487, "reward_judge_quality_mean": 0.7950000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.5656921863555908, "reward_total_composite_std": 0.3202142119407654} {"timestamp_utc": "2026-04-13T07:28:49Z", "mode": "train", "global_step": 54, "epoch": 0.005424409844299347, "loss": 0.0885, "grad_norm": 10.542903900146484, "learning_rate": 9.83939393939394e-06, "num_tokens": 100231.0, "completions/mean_length": 124.625, "completions/min_length": 86.0, "completions/max_length": 182.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 124.625, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 182.0, "rewards/meter/mean": 0.2036353051662445, "rewards/meter/std": 0.22631075978279114, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9939138293266296, "rewards/repeat_soft/std": 0.007315187249332666, "rewards/judge_quality/mean": 0.9200000166893005, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.38589751720428467, "rewards/total_composite/std": 0.2055649310350418, "reward": 0.38589751720428467, "reward_std": 0.2055649310350418, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.228908970952034, "sampling/sampling_logp_difference/max": 2.3076095581054688, "sampling/importance_sampling_ratio/min": 0.09949881583452225, "sampling/importance_sampling_ratio/mean": 1.044135332107544, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1012340635061264, "clip_ratio/low_mean": 0.13333098776638508, "clip_ratio/low_min": 0.13333098776638508, "clip_ratio/high_mean": 0.08108514919877052, "clip_ratio/high_max": 0.08108514919877052, "clip_ratio/region_mean": 0.2144161369651556, "reward_total_mean": 0.38589751720428467, "reward_meter_mean": 0.2036353051662445, "reward_meter_std": 0.22631075978279114, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9939138293266296, "reward_repeat_soft_std": 0.007315187249332666, "reward_judge_quality_mean": 0.9200000166893005, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.38589751720428467, "reward_total_composite_std": 0.2055649310350418} {"timestamp_utc": "2026-04-13T07:28:55Z", "mode": "train", "global_step": 55, "epoch": 0.0055248618784530384, "loss": 0.0139, "grad_norm": 20.996606826782227, "learning_rate": 9.836363636363637e-06, "num_tokens": 101772.0, "completions/mean_length": 25.625, "completions/min_length": 18.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.625, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.44943761825561523, "rewards/meter/std": 0.4661872088909149, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9460721015930176, "rewards/repeat_soft/std": 0.03294894099235535, "rewards/judge_quality/mean": 0.5600000023841858, "rewards/judge_quality/std": 0.22258226573467255, "rewards/total_composite/mean": 0.5090354681015015, "rewards/total_composite/std": 0.20502084493637085, "reward": 0.5090354681015015, "reward_std": 0.20502084493637085, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19217462837696075, "sampling/sampling_logp_difference/max": 1.8104991912841797, "sampling/importance_sampling_ratio/min": 0.16357247531414032, "sampling/importance_sampling_ratio/mean": 1.0293488502502441, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4901997074484825, "clip_ratio/low_mean": 0.06276315823197365, "clip_ratio/low_min": 0.06276315823197365, "clip_ratio/high_mean": 0.08341875951737165, "clip_ratio/high_max": 0.08341875951737165, "clip_ratio/region_mean": 0.1461819177493453, "reward_total_mean": 0.5090354681015015, "reward_meter_mean": 0.44943761825561523, "reward_meter_std": 0.4661872088909149, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9460721015930176, "reward_repeat_soft_std": 0.03294894099235535, "reward_judge_quality_mean": 0.5600000023841858, "reward_judge_quality_std": 0.22258226573467255, "reward_total_composite_mean": 0.5090354681015015, "reward_total_composite_std": 0.20502084493637085} {"timestamp_utc": "2026-04-13T07:29:02Z", "mode": "train", "global_step": 56, "epoch": 0.0056253139126067305, "loss": -0.0396, "grad_norm": 17.0747127532959, "learning_rate": 9.833333333333333e-06, "num_tokens": 103282.0, "completions/mean_length": 30.75, "completions/min_length": 24.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.75, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.7303705215454102, "rewards/meter/std": 0.4463520348072052, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.17136012017726898, "rewards/total_composite/mean": 0.5913689732551575, "rewards/total_composite/std": 0.18762490153312683, "reward": 0.5913689732551575, "reward_std": 0.18762490153312683, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18441377580165863, "sampling/sampling_logp_difference/max": 1.2306923866271973, "sampling/importance_sampling_ratio/min": 0.2920902669429779, "sampling/importance_sampling_ratio/mean": 1.0209691524505615, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4771666452288628, "clip_ratio/low_mean": 0.051488833501935005, "clip_ratio/low_min": 0.051488833501935005, "clip_ratio/high_mean": 0.12333367951214314, "clip_ratio/high_max": 0.12333367951214314, "clip_ratio/region_mean": 0.17482251301407814, "reward_total_mean": 0.5913689732551575, "reward_meter_mean": 0.7303705215454102, "reward_meter_std": 0.4463520348072052, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.17136012017726898, "reward_total_composite_mean": 0.5913689732551575, "reward_total_composite_std": 0.18762490153312683} {"timestamp_utc": "2026-04-13T07:29:09Z", "mode": "train", "global_step": 57, "epoch": 0.005725765946760422, "loss": 0.0667, "grad_norm": 9.708298683166504, "learning_rate": 9.830303030303032e-06, "num_tokens": 105531.0, "completions/mean_length": 111.125, "completions/min_length": 78.0, "completions/max_length": 153.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 111.125, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 153.0, "rewards/meter/mean": 0.425605833530426, "rewards/meter/std": 0.37622109055519104, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9927740693092346, "rewards/repeat_soft/std": 0.0072191013023257256, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.18539534509181976, "rewards/total_composite/mean": 0.4965977072715759, "rewards/total_composite/std": 0.1536581963300705, "reward": 0.4965977072715759, "reward_std": 0.1536581963300705, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19263683259487152, "sampling/sampling_logp_difference/max": 1.5666651725769043, "sampling/importance_sampling_ratio/min": 0.20874013006687164, "sampling/importance_sampling_ratio/mean": 1.0319360494613647, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.05468812584877, "clip_ratio/low_mean": 0.11199099197983742, "clip_ratio/low_min": 0.11199099197983742, "clip_ratio/high_mean": 0.058838874101638794, "clip_ratio/high_max": 0.058838874101638794, "clip_ratio/region_mean": 0.1708298660814762, "reward_total_mean": 0.4965977072715759, "reward_meter_mean": 0.425605833530426, "reward_meter_std": 0.37622109055519104, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9927740693092346, "reward_repeat_soft_std": 0.0072191013023257256, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.18539534509181976, "reward_total_composite_mean": 0.4965977072715759, "reward_total_composite_std": 0.1536581963300705} {"timestamp_utc": "2026-04-13T07:29:17Z", "mode": "train", "global_step": 58, "epoch": 0.005826217980914114, "loss": 0.1245, "grad_norm": 10.96164608001709, "learning_rate": 9.827272727272729e-06, "num_tokens": 107742.0, "completions/mean_length": 93.375, "completions/min_length": 58.0, "completions/max_length": 137.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.375, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 137.0, "rewards/meter/mean": 0.4870108962059021, "rewards/meter/std": 0.451114296913147, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.15430334210395813, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.99791419506073, "rewards/repeat_soft/std": 0.0024445788003504276, "rewards/judge_quality/mean": 0.5362499952316284, "rewards/judge_quality/std": 0.22385822236537933, "rewards/total_composite/mean": 0.44787633419036865, "rewards/total_composite/std": 0.2583901584148407, "reward": 0.44787633419036865, "reward_std": 0.2583901882171631, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22290265560150146, "sampling/sampling_logp_difference/max": 1.734273910522461, "sampling/importance_sampling_ratio/min": 0.17652831971645355, "sampling/importance_sampling_ratio/mean": 1.019986867904663, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.226851150393486, "clip_ratio/low_mean": 0.12805911153554916, "clip_ratio/low_min": 0.12805911153554916, "clip_ratio/high_mean": 0.08842651732265949, "clip_ratio/high_max": 0.08842651732265949, "clip_ratio/region_mean": 0.21648562885820866, "reward_total_mean": 0.44787633419036865, "reward_meter_mean": 0.4870108962059021, "reward_meter_std": 0.451114296913147, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.15430334210395813, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.99791419506073, "reward_repeat_soft_std": 0.0024445788003504276, "reward_judge_quality_mean": 0.5362499952316284, "reward_judge_quality_std": 0.22385822236537933, "reward_total_composite_mean": 0.44787633419036865, "reward_total_composite_std": 0.2583901584148407} {"timestamp_utc": "2026-04-13T07:29:26Z", "mode": "train", "global_step": 59, "epoch": 0.005926670015067805, "loss": 0.467, "grad_norm": 13.283112525939941, "learning_rate": 9.824242424242425e-06, "num_tokens": 109807.0, "completions/mean_length": 79.125, "completions/min_length": 34.0, "completions/max_length": 258.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.125, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 258.0, "rewards/meter/mean": 0.4121353030204773, "rewards/meter/std": 0.47703149914741516, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9690682888031006, "rewards/repeat_soft/std": 0.02924323081970215, "rewards/judge_quality/mean": 0.7362500429153442, "rewards/judge_quality/std": 0.25376805663108826, "rewards/total_composite/mean": 0.5556825399398804, "rewards/total_composite/std": 0.3225981891155243, "reward": 0.5556825399398804, "reward_std": 0.3225981891155243, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1868036389350891, "sampling/sampling_logp_difference/max": 1.6826361417770386, "sampling/importance_sampling_ratio/min": 0.1858833134174347, "sampling/importance_sampling_ratio/mean": 1.0397764444351196, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.721499502658844, "clip_ratio/low_mean": 0.12037735246121883, "clip_ratio/low_min": 0.12037735246121883, "clip_ratio/high_mean": 0.05528116412460804, "clip_ratio/high_max": 0.05528116412460804, "clip_ratio/region_mean": 0.17565851658582687, "reward_total_mean": 0.5556825399398804, "reward_meter_mean": 0.4121353030204773, "reward_meter_std": 0.47703149914741516, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9690682888031006, "reward_repeat_soft_std": 0.02924323081970215, "reward_judge_quality_mean": 0.7362500429153442, "reward_judge_quality_std": 0.25376805663108826, "reward_total_composite_mean": 0.5556825399398804, "reward_total_composite_std": 0.3225981891155243} {"timestamp_utc": "2026-04-13T07:29:33Z", "mode": "train", "global_step": 60, "epoch": 0.006027122049221497, "loss": 0.1308, "grad_norm": 16.130277633666992, "learning_rate": 9.821212121212122e-06, "num_tokens": 111362.0, "completions/mean_length": 49.375, "completions/min_length": 31.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.375, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.6151303052902222, "rewards/meter/std": 0.3767223656177521, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9902870655059814, "rewards/repeat_soft/std": 0.013324901461601257, "rewards/judge_quality/mean": 0.6850000023841858, "rewards/judge_quality/std": 0.22915996611118317, "rewards/total_composite/mean": 0.6230052709579468, "rewards/total_composite/std": 0.20295588672161102, "reward": 0.6230052709579468, "reward_std": 0.20295588672161102, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19441068172454834, "sampling/sampling_logp_difference/max": 1.3564529418945312, "sampling/importance_sampling_ratio/min": 0.2575727701187134, "sampling/importance_sampling_ratio/mean": 1.0002994537353516, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6784673035144806, "clip_ratio/low_mean": 0.12781431712210178, "clip_ratio/low_min": 0.12781431712210178, "clip_ratio/high_mean": 0.07399587146937847, "clip_ratio/high_max": 0.07399587146937847, "clip_ratio/region_mean": 0.20181018859148026, "reward_total_mean": 0.6230052709579468, "reward_meter_mean": 0.6151303052902222, "reward_meter_std": 0.3767223656177521, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9902870655059814, "reward_repeat_soft_std": 0.013324901461601257, "reward_judge_quality_mean": 0.6850000023841858, "reward_judge_quality_std": 0.22915996611118317, "reward_total_composite_mean": 0.6230052709579468, "reward_total_composite_std": 0.20295588672161102} {"timestamp_utc": "2026-04-13T07:29:40Z", "mode": "train", "global_step": 61, "epoch": 0.006127574083375188, "loss": 0.1142, "grad_norm": 13.33469295501709, "learning_rate": 9.81818181818182e-06, "num_tokens": 113192.0, "completions/mean_length": 54.75, "completions/min_length": 38.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.75, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.3869086503982544, "rewards/meter/std": 0.45730486512184143, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9924991130828857, "rewards/repeat_soft/std": 0.01105050090700388, "rewards/judge_quality/mean": 0.5062500238418579, "rewards/judge_quality/std": 0.13265827298164368, "rewards/total_composite/mean": 0.4687749743461609, "rewards/total_composite/std": 0.16751234233379364, "reward": 0.4687749743461609, "reward_std": 0.16751235723495483, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23125039041042328, "sampling/sampling_logp_difference/max": 1.4231446981430054, "sampling/importance_sampling_ratio/min": 0.24095511436462402, "sampling/importance_sampling_ratio/mean": 1.0310434103012085, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9869754761457443, "clip_ratio/low_mean": 0.10279995948076248, "clip_ratio/low_min": 0.10279995948076248, "clip_ratio/high_mean": 0.08435150422155857, "clip_ratio/high_max": 0.08435150422155857, "clip_ratio/region_mean": 0.18715146370232105, "reward_total_mean": 0.4687749743461609, "reward_meter_mean": 0.3869086503982544, "reward_meter_std": 0.45730486512184143, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9924991130828857, "reward_repeat_soft_std": 0.01105050090700388, "reward_judge_quality_mean": 0.5062500238418579, "reward_judge_quality_std": 0.13265827298164368, "reward_total_composite_mean": 0.4687749743461609, "reward_total_composite_std": 0.16751234233379364} {"timestamp_utc": "2026-04-13T07:29:52Z", "mode": "train", "global_step": 62, "epoch": 0.00622802611752888, "loss": -0.0608, "grad_norm": 6.91639518737793, "learning_rate": 9.815151515151516e-06, "num_tokens": 116000.0, "completions/mean_length": 211.0, "completions/min_length": 140.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 168.0, "completions/min_terminated_length": 140.0, "completions/max_terminated_length": 237.0, "rewards/meter/mean": 0.20621223747730255, "rewards/meter/std": 0.1470290869474411, "rewards/count_adherence/mean": 0.824999988079071, "rewards/count_adherence/std": 0.345377653837204, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9960790872573853, "rewards/repeat_soft/std": 0.004181248601526022, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.3168066143989563, "rewards/total_composite/std": 0.15409335494041443, "reward": 0.3168066143989563, "reward_std": 0.15409335494041443, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22920776903629303, "sampling/sampling_logp_difference/max": 3.835620164871216, "sampling/importance_sampling_ratio/min": 0.021587945520877838, "sampling/importance_sampling_ratio/mean": 1.0088270902633667, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5278817415237427, "clip_ratio/low_mean": 0.028145695105195045, "clip_ratio/low_min": 0.028145695105195045, "clip_ratio/high_mean": 0.1687558777630329, "clip_ratio/high_max": 0.1687558777630329, "clip_ratio/region_mean": 0.19690157286822796, "reward_total_mean": 0.3168066143989563, "reward_meter_mean": 0.20621223747730255, "reward_meter_std": 0.1470290869474411, "reward_count_adherence_mean": 0.824999988079071, "reward_count_adherence_std": 0.345377653837204, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9960790872573853, "reward_repeat_soft_std": 0.004181248601526022, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.3168066143989563, "reward_total_composite_std": 0.15409335494041443} {"timestamp_utc": "2026-04-13T07:30:00Z", "mode": "train", "global_step": 63, "epoch": 0.006328478151682571, "loss": 0.1668, "grad_norm": 29.546297073364258, "learning_rate": 9.812121212121212e-06, "num_tokens": 117890.0, "completions/mean_length": 75.25, "completions/min_length": 55.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.25, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.4134503901004791, "rewards/meter/std": 0.332876592874527, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9948821067810059, "rewards/repeat_soft/std": 0.008461784571409225, "rewards/judge_quality/mean": 0.5325000286102295, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.4275481104850769, "rewards/total_composite/std": 0.1968495398759842, "reward": 0.4275481104850769, "reward_std": 0.1968495398759842, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22210216522216797, "sampling/sampling_logp_difference/max": 3.0588936805725098, "sampling/importance_sampling_ratio/min": 0.04693959653377533, "sampling/importance_sampling_ratio/mean": 1.0094012022018433, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1671466045081615, "clip_ratio/low_mean": 0.07250726781785488, "clip_ratio/low_min": 0.07250726781785488, "clip_ratio/high_mean": 0.11376019939780235, "clip_ratio/high_max": 0.11376019939780235, "clip_ratio/region_mean": 0.18626746721565723, "reward_total_mean": 0.4275481104850769, "reward_meter_mean": 0.4134503901004791, "reward_meter_std": 0.332876592874527, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9948821067810059, "reward_repeat_soft_std": 0.008461784571409225, "reward_judge_quality_mean": 0.5325000286102295, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.4275481104850769, "reward_total_composite_std": 0.1968495398759842} {"timestamp_utc": "2026-04-13T07:30:08Z", "mode": "train", "global_step": 64, "epoch": 0.006428930185836263, "loss": 0.051, "grad_norm": 12.051981925964355, "learning_rate": 9.809090909090911e-06, "num_tokens": 120161.0, "completions/mean_length": 101.875, "completions/min_length": 83.0, "completions/max_length": 126.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 101.875, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 126.0, "rewards/meter/mean": 0.4644274115562439, "rewards/meter/std": 0.44741523265838623, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9956849813461304, "rewards/repeat_soft/std": 0.005578116048127413, "rewards/judge_quality/mean": 0.5325000286102295, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.5090116262435913, "rewards/total_composite/std": 0.1652180701494217, "reward": 0.5090116262435913, "reward_std": 0.1652180552482605, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22064253687858582, "sampling/sampling_logp_difference/max": 1.6844162940979004, "sampling/importance_sampling_ratio/min": 0.1855527013540268, "sampling/importance_sampling_ratio/mean": 1.0209674835205078, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0903439819812775, "clip_ratio/low_mean": 0.08959254249930382, "clip_ratio/low_min": 0.08959254249930382, "clip_ratio/high_mean": 0.11559330485761166, "clip_ratio/high_max": 0.11559330485761166, "clip_ratio/region_mean": 0.20518584735691547, "reward_total_mean": 0.5090116262435913, "reward_meter_mean": 0.4644274115562439, "reward_meter_std": 0.44741523265838623, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9956849813461304, "reward_repeat_soft_std": 0.005578116048127413, "reward_judge_quality_mean": 0.5325000286102295, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.5090116262435913, "reward_total_composite_std": 0.1652180701494217} {"timestamp_utc": "2026-04-13T07:30:15Z", "mode": "train", "global_step": 65, "epoch": 0.0065293822199899544, "loss": 0.1801, "grad_norm": 24.44002342224121, "learning_rate": 9.806060606060607e-06, "num_tokens": 121722.0, "completions/mean_length": 45.125, "completions/min_length": 31.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.125, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.5212327241897583, "rewards/meter/std": 0.48654648661613464, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9990687370300293, "rewards/repeat_soft/std": 0.0013255642261356115, "rewards/judge_quality/mean": 0.48625001311302185, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.41597363352775574, "rewards/total_composite/std": 0.2080499529838562, "reward": 0.41597363352775574, "reward_std": 0.2080499529838562, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.28085407614707947, "sampling/sampling_logp_difference/max": 2.827544689178467, "sampling/importance_sampling_ratio/min": 0.059157922863960266, "sampling/importance_sampling_ratio/mean": 1.0095726251602173, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9724387228488922, "clip_ratio/low_mean": 0.14638710487633944, "clip_ratio/low_min": 0.14638710487633944, "clip_ratio/high_mean": 0.08752177841961384, "clip_ratio/high_max": 0.08752177841961384, "clip_ratio/region_mean": 0.23390888329595327, "reward_total_mean": 0.41597363352775574, "reward_meter_mean": 0.5212327241897583, "reward_meter_std": 0.48654648661613464, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9990687370300293, "reward_repeat_soft_std": 0.0013255642261356115, "reward_judge_quality_mean": 0.48625001311302185, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.41597363352775574, "reward_total_composite_std": 0.2080499529838562} {"timestamp_utc": "2026-04-13T07:30:26Z", "mode": "train", "global_step": 66, "epoch": 0.0066298342541436465, "loss": 0.0828, "grad_norm": 6.262032985687256, "learning_rate": 9.803030303030304e-06, "num_tokens": 123325.0, "completions/mean_length": 120.375, "completions/min_length": 33.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 64.42857360839844, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 142.0, "rewards/meter/mean": 0.5316421985626221, "rewards/meter/std": 0.3480125069618225, "rewards/count_adherence/mean": 0.6875, "rewards/count_adherence/std": 0.45806270837783813, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9995944499969482, "rewards/repeat_soft/std": 0.0009662444936111569, "rewards/judge_quality/mean": 0.4437500238418579, "rewards/judge_quality/std": 0.19935163855552673, "rewards/total_composite/mean": 0.4721200466156006, "rewards/total_composite/std": 0.24633915722370148, "reward": 0.4721200466156006, "reward_std": 0.24633914232254028, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22367364168167114, "sampling/sampling_logp_difference/max": 1.7374200820922852, "sampling/importance_sampling_ratio/min": 0.1759738177061081, "sampling/importance_sampling_ratio/mean": 1.0281990766525269, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8580587804317474, "clip_ratio/low_mean": 0.08735784329473972, "clip_ratio/low_min": 0.08735784329473972, "clip_ratio/high_mean": 0.09469257295131683, "clip_ratio/high_max": 0.09469257295131683, "clip_ratio/region_mean": 0.18205041624605656, "reward_total_mean": 0.4721200466156006, "reward_meter_mean": 0.5316421985626221, "reward_meter_std": 0.3480125069618225, "reward_count_adherence_mean": 0.6875, "reward_count_adherence_std": 0.45806270837783813, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9995944499969482, "reward_repeat_soft_std": 0.0009662444936111569, "reward_judge_quality_mean": 0.4437500238418579, "reward_judge_quality_std": 0.19935163855552673, "reward_total_composite_mean": 0.4721200466156006, "reward_total_composite_std": 0.24633915722370148} {"timestamp_utc": "2026-04-13T07:30:34Z", "mode": "train", "global_step": 67, "epoch": 0.006730286288297338, "loss": -0.0164, "grad_norm": 14.978940963745117, "learning_rate": 9.800000000000001e-06, "num_tokens": 125078.0, "completions/mean_length": 60.125, "completions/min_length": 40.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.125, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.5190232396125793, "rewards/meter/std": 0.41173845529556274, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.3720119297504425, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9963045716285706, "rewards/repeat_soft/std": 0.006186562590301037, "rewards/judge_quality/mean": 0.4762499928474426, "rewards/judge_quality/std": 0.19167961180210114, "rewards/total_composite/mean": 0.41957348585128784, "rewards/total_composite/std": 0.252689391374588, "reward": 0.41957348585128784, "reward_std": 0.252689391374588, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21659037470817566, "sampling/sampling_logp_difference/max": 1.5980007648468018, "sampling/importance_sampling_ratio/min": 0.202300563454628, "sampling/importance_sampling_ratio/mean": 1.0314173698425293, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.830552101135254, "clip_ratio/low_mean": 0.1511866319924593, "clip_ratio/low_min": 0.1511866319924593, "clip_ratio/high_mean": 0.04352152161300182, "clip_ratio/high_max": 0.04352152161300182, "clip_ratio/region_mean": 0.19470815360546112, "reward_total_mean": 0.41957348585128784, "reward_meter_mean": 0.5190232396125793, "reward_meter_std": 0.41173845529556274, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.3720119297504425, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9963045716285706, "reward_repeat_soft_std": 0.006186562590301037, "reward_judge_quality_mean": 0.4762499928474426, "reward_judge_quality_std": 0.19167961180210114, "reward_total_composite_mean": 0.41957348585128784, "reward_total_composite_std": 0.252689391374588} {"timestamp_utc": "2026-04-13T07:30:41Z", "mode": "train", "global_step": 68, "epoch": 0.00683073832245103, "loss": 0.1232, "grad_norm": 23.07606315612793, "learning_rate": 9.796969696969698e-06, "num_tokens": 126637.0, "completions/mean_length": 44.875, "completions/min_length": 39.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.875, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.5100425481796265, "rewards/meter/std": 0.3832434415817261, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.975085973739624, "rewards/repeat_soft/std": 0.027266472578048706, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.49085086584091187, "rewards/total_composite/std": 0.10883072763681412, "reward": 0.49085086584091187, "reward_std": 0.10883072018623352, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21352910995483398, "sampling/sampling_logp_difference/max": 2.5040464401245117, "sampling/importance_sampling_ratio/min": 0.08175352215766907, "sampling/importance_sampling_ratio/mean": 1.0196813344955444, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.520227499306202, "clip_ratio/low_mean": 0.09456020221114159, "clip_ratio/low_min": 0.09456020221114159, "clip_ratio/high_mean": 0.09431146644055843, "clip_ratio/high_max": 0.09431146644055843, "clip_ratio/region_mean": 0.18887166865170002, "reward_total_mean": 0.49085086584091187, "reward_meter_mean": 0.5100425481796265, "reward_meter_std": 0.3832434415817261, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.975085973739624, "reward_repeat_soft_std": 0.027266472578048706, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.49085086584091187, "reward_total_composite_std": 0.10883072763681412} {"timestamp_utc": "2026-04-13T07:30:48Z", "mode": "train", "global_step": 69, "epoch": 0.006931190356604721, "loss": 0.0529, "grad_norm": 22.51520347595215, "learning_rate": 9.793939393939394e-06, "num_tokens": 128609.0, "completions/mean_length": 59.5, "completions/min_length": 47.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.5, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.3263697326183319, "rewards/meter/std": 0.34783032536506653, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.988911509513855, "rewards/repeat_soft/std": 0.010033752769231796, "rewards/judge_quality/mean": 0.5362499952316284, "rewards/judge_quality/std": 0.1524970978498459, "rewards/total_composite/mean": 0.3768683075904846, "rewards/total_composite/std": 0.2702862620353699, "reward": 0.3768683075904846, "reward_std": 0.2702862620353699, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2645116448402405, "sampling/sampling_logp_difference/max": 2.039849281311035, "sampling/importance_sampling_ratio/min": 0.13004830479621887, "sampling/importance_sampling_ratio/mean": 1.003286361694336, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4887580871582031, "clip_ratio/low_mean": 0.1026343535631895, "clip_ratio/low_min": 0.1026343535631895, "clip_ratio/high_mean": 0.14072966761887074, "clip_ratio/high_max": 0.14072966761887074, "clip_ratio/region_mean": 0.24336402118206024, "reward_total_mean": 0.3768683075904846, "reward_meter_mean": 0.3263697326183319, "reward_meter_std": 0.34783032536506653, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.988911509513855, "reward_repeat_soft_std": 0.010033752769231796, "reward_judge_quality_mean": 0.5362499952316284, "reward_judge_quality_std": 0.1524970978498459, "reward_total_composite_mean": 0.3768683075904846, "reward_total_composite_std": 0.2702862620353699} {"timestamp_utc": "2026-04-13T07:30:56Z", "mode": "train", "global_step": 70, "epoch": 0.007031642390758413, "loss": -0.0466, "grad_norm": 12.199811935424805, "learning_rate": 9.790909090909093e-06, "num_tokens": 130737.0, "completions/mean_length": 94.0, "completions/min_length": 65.0, "completions/max_length": 131.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.0, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.2902809977531433, "rewards/meter/std": 0.12799283862113953, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9939800500869751, "rewards/repeat_soft/std": 0.004366800654679537, "rewards/judge_quality/mean": 0.38999998569488525, "rewards/judge_quality/std": 0.0975411981344223, "rewards/total_composite/mean": 0.4276006817817688, "rewards/total_composite/std": 0.03887045010924339, "reward": 0.4276006817817688, "reward_std": 0.03887045010924339, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.239035964012146, "sampling/sampling_logp_difference/max": 2.181110382080078, "sampling/importance_sampling_ratio/min": 0.11291608214378357, "sampling/importance_sampling_ratio/mean": 1.0222389698028564, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2758661061525345, "clip_ratio/low_mean": 0.10574052855372429, "clip_ratio/low_min": 0.10574052855372429, "clip_ratio/high_mean": 0.10558660514652729, "clip_ratio/high_max": 0.10558660514652729, "clip_ratio/region_mean": 0.21132713370025158, "reward_total_mean": 0.4276006817817688, "reward_meter_mean": 0.2902809977531433, "reward_meter_std": 0.12799283862113953, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9939800500869751, "reward_repeat_soft_std": 0.004366800654679537, "reward_judge_quality_mean": 0.38999998569488525, "reward_judge_quality_std": 0.0975411981344223, "reward_total_composite_mean": 0.4276006817817688, "reward_total_composite_std": 0.03887045010924339} {"timestamp_utc": "2026-04-13T07:31:02Z", "mode": "train", "global_step": 71, "epoch": 0.007132094424912104, "loss": 0.0143, "grad_norm": 14.534868240356445, "learning_rate": 9.787878787878788e-06, "num_tokens": 132332.0, "completions/mean_length": 27.375, "completions/min_length": 26.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.375, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.980559229850769, "rewards/meter/std": 0.01318947970867157, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9953598380088806, "rewards/repeat_soft/std": 0.0018749026348814368, "rewards/judge_quality/mean": 0.9200000166893005, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.9356783628463745, "rewards/total_composite/std": 0.007609147112816572, "reward": 0.9356783628463745, "reward_std": 0.007609153166413307, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06976129859685898, "sampling/sampling_logp_difference/max": 0.8716723918914795, "sampling/importance_sampling_ratio/min": 0.41825151443481445, "sampling/importance_sampling_ratio/mean": 1.0073102712631226, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.42719393596053123, "clip_ratio/low_mean": 0.022321429569274187, "clip_ratio/low_min": 0.022321429569274187, "clip_ratio/high_mean": 0.02252229768782854, "clip_ratio/high_max": 0.02252229768782854, "clip_ratio/region_mean": 0.04484372725710273, "reward_total_mean": 0.9356783628463745, "reward_meter_mean": 0.980559229850769, "reward_meter_std": 0.01318947970867157, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9953598380088806, "reward_repeat_soft_std": 0.0018749026348814368, "reward_judge_quality_mean": 0.9200000166893005, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.9356783628463745, "reward_total_composite_std": 0.007609147112816572} {"timestamp_utc": "2026-04-13T07:31:09Z", "mode": "train", "global_step": 72, "epoch": 0.007232546459065796, "loss": 0.0149, "grad_norm": 18.851024627685547, "learning_rate": 9.784848484848486e-06, "num_tokens": 134250.0, "completions/mean_length": 77.75, "completions/min_length": 65.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.75, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.5707041025161743, "rewards/meter/std": 0.3704380989074707, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9915953874588013, "rewards/repeat_soft/std": 0.003521057078614831, "rewards/judge_quality/mean": 0.6449999809265137, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.579657256603241, "rewards/total_composite/std": 0.16086405515670776, "reward": 0.579657256603241, "reward_std": 0.16086405515670776, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15077872574329376, "sampling/sampling_logp_difference/max": 1.8281669616699219, "sampling/importance_sampling_ratio/min": 0.16070789098739624, "sampling/importance_sampling_ratio/mean": 1.001753330230713, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7165041640400887, "clip_ratio/low_mean": 0.077230763155967, "clip_ratio/low_min": 0.077230763155967, "clip_ratio/high_mean": 0.03872744762338698, "clip_ratio/high_max": 0.03872744762338698, "clip_ratio/region_mean": 0.11595821077935398, "reward_total_mean": 0.579657256603241, "reward_meter_mean": 0.5707041025161743, "reward_meter_std": 0.3704380989074707, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9915953874588013, "reward_repeat_soft_std": 0.003521057078614831, "reward_judge_quality_mean": 0.6449999809265137, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.579657256603241, "reward_total_composite_std": 0.16086405515670776} {"timestamp_utc": "2026-04-13T07:31:16Z", "mode": "train", "global_step": 73, "epoch": 0.007332998493219488, "loss": 0.0722, "grad_norm": 16.596925735473633, "learning_rate": 9.781818181818183e-06, "num_tokens": 135829.0, "completions/mean_length": 38.375, "completions/min_length": 33.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.375, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.7099935412406921, "rewards/meter/std": 0.3367968797683716, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9981318712234497, "rewards/repeat_soft/std": 0.005283824168145657, "rewards/judge_quality/mean": 0.5862500071525574, "rewards/judge_quality/std": 0.22984081506729126, "rewards/total_composite/mean": 0.6462399959564209, "rewards/total_composite/std": 0.21094709634780884, "reward": 0.6462399959564209, "reward_std": 0.21094708144664764, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1958298534154892, "sampling/sampling_logp_difference/max": 1.196608543395996, "sampling/importance_sampling_ratio/min": 0.30221742391586304, "sampling/importance_sampling_ratio/mean": 1.0368590354919434, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8273597061634064, "clip_ratio/low_mean": 0.13772443495690823, "clip_ratio/low_min": 0.13772443495690823, "clip_ratio/high_mean": 0.09337121434509754, "clip_ratio/high_max": 0.09337121434509754, "clip_ratio/region_mean": 0.23109564930200577, "reward_total_mean": 0.6462399959564209, "reward_meter_mean": 0.7099935412406921, "reward_meter_std": 0.3367968797683716, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9981318712234497, "reward_repeat_soft_std": 0.005283824168145657, "reward_judge_quality_mean": 0.5862500071525574, "reward_judge_quality_std": 0.22984081506729126, "reward_total_composite_mean": 0.6462399959564209, "reward_total_composite_std": 0.21094709634780884} {"timestamp_utc": "2026-04-13T07:31:23Z", "mode": "train", "global_step": 74, "epoch": 0.007433450527373179, "loss": 0.0166, "grad_norm": 16.803701400756836, "learning_rate": 9.77878787878788e-06, "num_tokens": 137366.0, "completions/mean_length": 46.125, "completions/min_length": 31.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.125, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.6728395819664001, "rewards/meter/std": 0.34131065011024475, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9953755736351013, "rewards/repeat_soft/std": 0.004812015686184168, "rewards/judge_quality/mean": 0.9275000095367432, "rewards/judge_quality/std": 0.013887288980185986, "rewards/total_composite/mean": 0.756302535533905, "rewards/total_composite/std": 0.20829513669013977, "reward": 0.756302535533905, "reward_std": 0.20829510688781738, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12959113717079163, "sampling/sampling_logp_difference/max": 1.2154483795166016, "sampling/importance_sampling_ratio/min": 0.29657700657844543, "sampling/importance_sampling_ratio/mean": 1.0163564682006836, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8411549031734467, "clip_ratio/low_mean": 0.06045190338045359, "clip_ratio/low_min": 0.06045190338045359, "clip_ratio/high_mean": 0.05965705122798681, "clip_ratio/high_max": 0.05965705122798681, "clip_ratio/region_mean": 0.1201089546084404, "reward_total_mean": 0.756302535533905, "reward_meter_mean": 0.6728395819664001, "reward_meter_std": 0.34131065011024475, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9953755736351013, "reward_repeat_soft_std": 0.004812015686184168, "reward_judge_quality_mean": 0.9275000095367432, "reward_judge_quality_std": 0.013887288980185986, "reward_total_composite_mean": 0.756302535533905, "reward_total_composite_std": 0.20829513669013977} {"timestamp_utc": "2026-04-13T07:31:32Z", "mode": "train", "global_step": 75, "epoch": 0.007533902561526871, "loss": 0.3489, "grad_norm": 11.72706127166748, "learning_rate": 9.775757575757576e-06, "num_tokens": 139715.0, "completions/mean_length": 119.625, "completions/min_length": 79.0, "completions/max_length": 264.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 119.625, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 264.0, "rewards/meter/mean": 0.7541071772575378, "rewards/meter/std": 0.2662733495235443, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.3720119297504425, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9913018941879272, "rewards/repeat_soft/std": 0.011900298297405243, "rewards/judge_quality/mean": 0.5824999809265137, "rewards/judge_quality/std": 0.176776722073555, "rewards/total_composite/mean": 0.606920599937439, "rewards/total_composite/std": 0.20927901566028595, "reward": 0.606920599937439, "reward_std": 0.20927901566028595, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2122229039669037, "sampling/sampling_logp_difference/max": 1.480539321899414, "sampling/importance_sampling_ratio/min": 0.22751495242118835, "sampling/importance_sampling_ratio/mean": 1.0367655754089355, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.362242892384529, "clip_ratio/low_mean": 0.1122982744127512, "clip_ratio/low_min": 0.1122982744127512, "clip_ratio/high_mean": 0.07517952844500542, "clip_ratio/high_max": 0.07517952844500542, "clip_ratio/region_mean": 0.18747780285775661, "reward_total_mean": 0.606920599937439, "reward_meter_mean": 0.7541071772575378, "reward_meter_std": 0.2662733495235443, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.3720119297504425, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9913018941879272, "reward_repeat_soft_std": 0.011900298297405243, "reward_judge_quality_mean": 0.5824999809265137, "reward_judge_quality_std": 0.176776722073555, "reward_total_composite_mean": 0.606920599937439, "reward_total_composite_std": 0.20927901566028595} {"timestamp_utc": "2026-04-13T07:31:39Z", "mode": "train", "global_step": 76, "epoch": 0.0076343545956805625, "loss": 0.0372, "grad_norm": 18.240943908691406, "learning_rate": 9.772727272727273e-06, "num_tokens": 141197.0, "completions/mean_length": 31.25, "completions/min_length": 18.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.25, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.6227948069572449, "rewards/meter/std": 0.4872068166732788, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9694886207580566, "rewards/repeat_soft/std": 0.019950784742832184, "rewards/judge_quality/mean": 0.5275000333786011, "rewards/judge_quality/std": 0.2499571591615677, "rewards/total_composite/mean": 0.5201561450958252, "rewards/total_composite/std": 0.23341584205627441, "reward": 0.5201561450958252, "reward_std": 0.2334158569574356, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19917994737625122, "sampling/sampling_logp_difference/max": 1.2310771942138672, "sampling/importance_sampling_ratio/min": 0.2919778823852539, "sampling/importance_sampling_ratio/mean": 1.0508776903152466, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.957547053694725, "clip_ratio/low_mean": 0.062437155516818166, "clip_ratio/low_min": 0.062437155516818166, "clip_ratio/high_mean": 0.10055694729089737, "clip_ratio/high_max": 0.10055694729089737, "clip_ratio/region_mean": 0.16299410280771554, "reward_total_mean": 0.5201561450958252, "reward_meter_mean": 0.6227948069572449, "reward_meter_std": 0.4872068166732788, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9694886207580566, "reward_repeat_soft_std": 0.019950784742832184, "reward_judge_quality_mean": 0.5275000333786011, "reward_judge_quality_std": 0.2499571591615677, "reward_total_composite_mean": 0.5201561450958252, "reward_total_composite_std": 0.23341584205627441} {"timestamp_utc": "2026-04-13T07:31:46Z", "mode": "train", "global_step": 77, "epoch": 0.0077348066298342545, "loss": 0.132, "grad_norm": 16.104795455932617, "learning_rate": 9.76969696969697e-06, "num_tokens": 142809.0, "completions/mean_length": 45.5, "completions/min_length": 32.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.5, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.5599648952484131, "rewards/meter/std": 0.4068310260772705, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9915940761566162, "rewards/repeat_soft/std": 0.012580431066453457, "rewards/judge_quality/mean": 0.5887500047683716, "rewards/judge_quality/std": 0.22699514031410217, "rewards/total_composite/mean": 0.5684210658073425, "rewards/total_composite/std": 0.18888187408447266, "reward": 0.5684210658073425, "reward_std": 0.18888187408447266, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.200801819562912, "sampling/sampling_logp_difference/max": 3.130115509033203, "sampling/importance_sampling_ratio/min": 0.04371274635195732, "sampling/importance_sampling_ratio/mean": 1.0195224285125732, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2177944853901863, "clip_ratio/low_mean": 0.08189102727919817, "clip_ratio/low_min": 0.08189102727919817, "clip_ratio/high_mean": 0.09817468374967575, "clip_ratio/high_max": 0.09817468374967575, "clip_ratio/region_mean": 0.18006571102887392, "reward_total_mean": 0.5684210658073425, "reward_meter_mean": 0.5599648952484131, "reward_meter_std": 0.4068310260772705, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9915940761566162, "reward_repeat_soft_std": 0.012580431066453457, "reward_judge_quality_mean": 0.5887500047683716, "reward_judge_quality_std": 0.22699514031410217, "reward_total_composite_mean": 0.5684210658073425, "reward_total_composite_std": 0.18888187408447266} {"timestamp_utc": "2026-04-13T07:31:53Z", "mode": "train", "global_step": 78, "epoch": 0.007835258663987946, "loss": 0.1556, "grad_norm": 9.579455375671387, "learning_rate": 9.766666666666667e-06, "num_tokens": 145314.0, "completions/mean_length": 118.125, "completions/min_length": 70.0, "completions/max_length": 137.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 118.125, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 137.0, "rewards/meter/mean": 0.6192508935928345, "rewards/meter/std": 0.2793404459953308, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.996671199798584, "rewards/repeat_soft/std": 0.00344731449149549, "rewards/judge_quality/mean": 0.6074999570846558, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.60548335313797, "rewards/total_composite/std": 0.1651448756456375, "reward": 0.60548335313797, "reward_std": 0.1651448756456375, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2017233520746231, "sampling/sampling_logp_difference/max": 2.272388458251953, "sampling/importance_sampling_ratio/min": 0.10306572169065475, "sampling/importance_sampling_ratio/mean": 1.0252689123153687, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0700202882289886, "clip_ratio/low_mean": 0.09481169655919075, "clip_ratio/low_min": 0.09481169655919075, "clip_ratio/high_mean": 0.10964499041438103, "clip_ratio/high_max": 0.10964499041438103, "clip_ratio/region_mean": 0.20445668697357178, "reward_total_mean": 0.60548335313797, "reward_meter_mean": 0.6192508935928345, "reward_meter_std": 0.2793404459953308, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.996671199798584, "reward_repeat_soft_std": 0.00344731449149549, "reward_judge_quality_mean": 0.6074999570846558, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.60548335313797, "reward_total_composite_std": 0.1651448756456375} {"timestamp_utc": "2026-04-13T07:32:06Z", "mode": "train", "global_step": 79, "epoch": 0.007935710698141637, "loss": -0.1215, "grad_norm": 5.458466529846191, "learning_rate": 9.763636363636365e-06, "num_tokens": 147338.0, "completions/mean_length": 140.0, "completions/min_length": 46.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 86.85714721679688, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.2900845408439636, "rewards/meter/std": 0.2756892442703247, "rewards/count_adherence/mean": 0.8333333730697632, "rewards/count_adherence/std": 0.3563483655452728, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9936281442642212, "rewards/repeat_soft/std": 0.0072111692279577255, "rewards/judge_quality/mean": 0.5737500190734863, "rewards/judge_quality/std": 0.3009004592895508, "rewards/total_composite/mean": 0.44272592663764954, "rewards/total_composite/std": 0.18702588975429535, "reward": 0.44272592663764954, "reward_std": 0.18702590465545654, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20270249247550964, "sampling/sampling_logp_difference/max": 2.109290599822998, "sampling/importance_sampling_ratio/min": 0.12132399529218674, "sampling/importance_sampling_ratio/mean": 1.0167397260665894, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4257397055625916, "clip_ratio/low_mean": 0.09781302325427532, "clip_ratio/low_min": 0.09781302325427532, "clip_ratio/high_mean": 0.07240715436637402, "clip_ratio/high_max": 0.07240715436637402, "clip_ratio/region_mean": 0.17022017762064934, "reward_total_mean": 0.44272592663764954, "reward_meter_mean": 0.2900845408439636, "reward_meter_std": 0.2756892442703247, "reward_count_adherence_mean": 0.8333333730697632, "reward_count_adherence_std": 0.3563483655452728, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9936281442642212, "reward_repeat_soft_std": 0.0072111692279577255, "reward_judge_quality_mean": 0.5737500190734863, "reward_judge_quality_std": 0.3009004592895508, "reward_total_composite_mean": 0.44272592663764954, "reward_total_composite_std": 0.18702588975429535} {"timestamp_utc": "2026-04-13T07:32:17Z", "mode": "train", "global_step": 80, "epoch": 0.00803616273229533, "loss": 0.6328, "grad_norm": 9.749089241027832, "learning_rate": 9.760606060606062e-06, "num_tokens": 149447.0, "completions/mean_length": 94.625, "completions/min_length": 33.0, "completions/max_length": 422.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.625, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 422.0, "rewards/meter/mean": 0.7609310746192932, "rewards/meter/std": 0.37621188163757324, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9743838310241699, "rewards/repeat_soft/std": 0.06604637950658798, "rewards/judge_quality/mean": 0.59375, "rewards/judge_quality/std": 0.22965426743030548, "rewards/total_composite/mean": 0.6037291288375854, "rewards/total_composite/std": 0.1989206224679947, "reward": 0.6037291288375854, "reward_std": 0.19892063736915588, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15118443965911865, "sampling/sampling_logp_difference/max": 1.5827999114990234, "sampling/importance_sampling_ratio/min": 0.20539918541908264, "sampling/importance_sampling_ratio/mean": 1.0348566770553589, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1621425449848175, "clip_ratio/low_mean": 0.06869752984493971, "clip_ratio/low_min": 0.06869752984493971, "clip_ratio/high_mean": 0.0752419987693429, "clip_ratio/high_max": 0.0752419987693429, "clip_ratio/region_mean": 0.1439395286142826, "reward_total_mean": 0.6037291288375854, "reward_meter_mean": 0.7609310746192932, "reward_meter_std": 0.37621188163757324, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9743838310241699, "reward_repeat_soft_std": 0.06604637950658798, "reward_judge_quality_mean": 0.59375, "reward_judge_quality_std": 0.22965426743030548, "reward_total_composite_mean": 0.6037291288375854, "reward_total_composite_std": 0.1989206224679947} {"timestamp_utc": "2026-04-13T07:32:26Z", "mode": "train", "global_step": 81, "epoch": 0.008136614766449021, "loss": 0.2858, "grad_norm": 9.317183494567871, "learning_rate": 9.757575757575758e-06, "num_tokens": 152215.0, "completions/mean_length": 136.0, "completions/min_length": 75.0, "completions/max_length": 226.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 136.0, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 226.0, "rewards/meter/mean": 0.8717004060745239, "rewards/meter/std": 0.18755942583084106, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9913686513900757, "rewards/repeat_soft/std": 0.01117410883307457, "rewards/judge_quality/mean": 0.5112500190734863, "rewards/judge_quality/std": 0.18216457962989807, "rewards/total_composite/mean": 0.643903374671936, "rewards/total_composite/std": 0.1500636488199234, "reward": 0.643903374671936, "reward_std": 0.1500636488199234, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2290331870317459, "sampling/sampling_logp_difference/max": 1.9447827339172363, "sampling/importance_sampling_ratio/min": 0.14301829040050507, "sampling/importance_sampling_ratio/mean": 1.0455464124679565, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6830307990312576, "clip_ratio/low_mean": 0.12898291274905205, "clip_ratio/low_min": 0.12898291274905205, "clip_ratio/high_mean": 0.0660889595746994, "clip_ratio/high_max": 0.0660889595746994, "clip_ratio/region_mean": 0.19507187232375145, "reward_total_mean": 0.643903374671936, "reward_meter_mean": 0.8717004060745239, "reward_meter_std": 0.18755942583084106, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9913686513900757, "reward_repeat_soft_std": 0.01117410883307457, "reward_judge_quality_mean": 0.5112500190734863, "reward_judge_quality_std": 0.18216457962989807, "reward_total_composite_mean": 0.643903374671936, "reward_total_composite_std": 0.1500636488199234} {"timestamp_utc": "2026-04-13T07:32:33Z", "mode": "train", "global_step": 82, "epoch": 0.008237066800602712, "loss": 0.2129, "grad_norm": 18.911800384521484, "learning_rate": 9.754545454545455e-06, "num_tokens": 153757.0, "completions/mean_length": 43.75, "completions/min_length": 33.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.75, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.3292384743690491, "rewards/meter/std": 0.38146013021469116, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9974339008331299, "rewards/repeat_soft/std": 0.004254767205566168, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.4292713403701782, "rewards/total_composite/std": 0.1198873519897461, "reward": 0.4292713403701782, "reward_std": 0.1198873519897461, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22892795503139496, "sampling/sampling_logp_difference/max": 1.2771120071411133, "sampling/importance_sampling_ratio/min": 0.27884143590927124, "sampling/importance_sampling_ratio/mean": 1.0563212633132935, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1919937133789062, "clip_ratio/low_mean": 0.16997694596648216, "clip_ratio/low_min": 0.16997694596648216, "clip_ratio/high_mean": 0.048161765560507774, "clip_ratio/high_max": 0.048161765560507774, "clip_ratio/region_mean": 0.21813871152698994, "reward_total_mean": 0.4292713403701782, "reward_meter_mean": 0.3292384743690491, "reward_meter_std": 0.38146013021469116, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9974339008331299, "reward_repeat_soft_std": 0.004254767205566168, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.4292713403701782, "reward_total_composite_std": 0.1198873519897461} {"timestamp_utc": "2026-04-13T07:32:40Z", "mode": "train", "global_step": 83, "epoch": 0.008337518834756403, "loss": 0.0794, "grad_norm": 12.305427551269531, "learning_rate": 9.751515151515152e-06, "num_tokens": 155299.0, "completions/mean_length": 47.75, "completions/min_length": 37.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.75, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.7958900928497314, "rewards/meter/std": 0.3201044499874115, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9813162088394165, "rewards/repeat_soft/std": 0.050224728882312775, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.6428062915802002, "rewards/total_composite/std": 0.19279375672340393, "reward": 0.6428062915802002, "reward_std": 0.19279374182224274, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18508756160736084, "sampling/sampling_logp_difference/max": 1.5233378410339355, "sampling/importance_sampling_ratio/min": 0.21798309683799744, "sampling/importance_sampling_ratio/mean": 1.0269668102264404, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6632934510707855, "clip_ratio/low_mean": 0.11185488197952509, "clip_ratio/low_min": 0.11185488197952509, "clip_ratio/high_mean": 0.045780474320054054, "clip_ratio/high_max": 0.045780474320054054, "clip_ratio/region_mean": 0.15763535629957914, "reward_total_mean": 0.6428062915802002, "reward_meter_mean": 0.7958900928497314, "reward_meter_std": 0.3201044499874115, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9813162088394165, "reward_repeat_soft_std": 0.050224728882312775, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.6428062915802002, "reward_total_composite_std": 0.19279375672340393} {"timestamp_utc": "2026-04-13T07:32:48Z", "mode": "train", "global_step": 84, "epoch": 0.008437970868910096, "loss": 0.0305, "grad_norm": 15.982399940490723, "learning_rate": 9.74848484848485e-06, "num_tokens": 157087.0, "completions/mean_length": 42.5, "completions/min_length": 34.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.5, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.39095914363861084, "rewards/meter/std": 0.42909476161003113, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9967079758644104, "rewards/repeat_soft/std": 0.008959452621638775, "rewards/judge_quality/mean": 0.3812499940395355, "rewards/judge_quality/std": 0.08166787773370743, "rewards/total_composite/mean": 0.4579356908798218, "rewards/total_composite/std": 0.1210685521364212, "reward": 0.4579356908798218, "reward_std": 0.1210685521364212, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24277810752391815, "sampling/sampling_logp_difference/max": 1.4623289108276367, "sampling/importance_sampling_ratio/min": 0.23169603943824768, "sampling/importance_sampling_ratio/mean": 1.0521039962768555, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.471090257167816, "clip_ratio/low_mean": 0.1764454822987318, "clip_ratio/low_min": 0.1764454822987318, "clip_ratio/high_mean": 0.08806561306118965, "clip_ratio/high_max": 0.08806561306118965, "clip_ratio/region_mean": 0.26451109535992146, "reward_total_mean": 0.4579356908798218, "reward_meter_mean": 0.39095914363861084, "reward_meter_std": 0.42909476161003113, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9967079758644104, "reward_repeat_soft_std": 0.008959452621638775, "reward_judge_quality_mean": 0.3812499940395355, "reward_judge_quality_std": 0.08166787773370743, "reward_total_composite_mean": 0.4579356908798218, "reward_total_composite_std": 0.1210685521364212} {"timestamp_utc": "2026-04-13T07:32:55Z", "mode": "train", "global_step": 85, "epoch": 0.008538422903063787, "loss": -0.0612, "grad_norm": 28.472885131835938, "learning_rate": 9.745454545454547e-06, "num_tokens": 158426.0, "completions/mean_length": 22.375, "completions/min_length": 17.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.375, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.6606827974319458, "rewards/meter/std": 0.4175296425819397, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.930449366569519, "rewards/repeat_soft/std": 0.06305917352437973, "rewards/judge_quality/mean": 0.4025000035762787, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.5210764408111572, "rewards/total_composite/std": 0.11448659002780914, "reward": 0.5210764408111572, "reward_std": 0.11448659002780914, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22843873500823975, "sampling/sampling_logp_difference/max": 2.0268335342407227, "sampling/importance_sampling_ratio/min": 0.13175204396247864, "sampling/importance_sampling_ratio/mean": 1.0153146982192993, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6977891623973846, "clip_ratio/low_mean": 0.08766401372849941, "clip_ratio/low_min": 0.08766401372849941, "clip_ratio/high_mean": 0.11823012121021748, "clip_ratio/high_max": 0.11823012121021748, "clip_ratio/region_mean": 0.2058941349387169, "reward_total_mean": 0.5210764408111572, "reward_meter_mean": 0.6606827974319458, "reward_meter_std": 0.4175296425819397, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.930449366569519, "reward_repeat_soft_std": 0.06305917352437973, "reward_judge_quality_mean": 0.4025000035762787, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.5210764408111572, "reward_total_composite_std": 0.11448659002780914} {"timestamp_utc": "2026-04-13T07:33:03Z", "mode": "train", "global_step": 86, "epoch": 0.008638874937217478, "loss": 0.0581, "grad_norm": 19.20193862915039, "learning_rate": 9.742424242424244e-06, "num_tokens": 160546.0, "completions/mean_length": 86.0, "completions/min_length": 67.0, "completions/max_length": 135.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 135.0, "rewards/meter/mean": 0.24867114424705505, "rewards/meter/std": 0.20864476263523102, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9985886216163635, "rewards/repeat_soft/std": 0.0018666742835193872, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.10966669768095016, "rewards/total_composite/mean": 0.41330647468566895, "rewards/total_composite/std": 0.06062160059809685, "reward": 0.41330647468566895, "reward_std": 0.06062160059809685, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23355013132095337, "sampling/sampling_logp_difference/max": 3.66347599029541, "sampling/importance_sampling_ratio/min": 0.02564322203397751, "sampling/importance_sampling_ratio/mean": 1.0034799575805664, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.218077875673771, "clip_ratio/low_mean": 0.12993094697594643, "clip_ratio/low_min": 0.12993094697594643, "clip_ratio/high_mean": 0.059350667521357536, "clip_ratio/high_max": 0.059350667521357536, "clip_ratio/region_mean": 0.18928161449730396, "reward_total_mean": 0.41330647468566895, "reward_meter_mean": 0.24867114424705505, "reward_meter_std": 0.20864476263523102, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9985886216163635, "reward_repeat_soft_std": 0.0018666742835193872, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.10966669768095016, "reward_total_composite_mean": 0.41330647468566895, "reward_total_composite_std": 0.06062160059809685} {"timestamp_utc": "2026-04-13T07:33:10Z", "mode": "train", "global_step": 87, "epoch": 0.00873932697137117, "loss": 0.145, "grad_norm": 23.893922805786133, "learning_rate": 9.739393939393941e-06, "num_tokens": 161860.0, "completions/mean_length": 23.25, "completions/min_length": 15.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.25, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.2548510432243347, "rewards/meter/std": 0.3831472098827362, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9499562978744507, "rewards/repeat_soft/std": 0.021399270743131638, "rewards/judge_quality/mean": 0.5600000023841858, "rewards/judge_quality/std": 0.22258226573467255, "rewards/total_composite/mean": 0.4158652424812317, "rewards/total_composite/std": 0.10863844305276871, "reward": 0.4158652424812317, "reward_std": 0.10863843560218811, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21412859857082367, "sampling/sampling_logp_difference/max": 1.3826007843017578, "sampling/importance_sampling_ratio/min": 0.25092509388923645, "sampling/importance_sampling_ratio/mean": 1.0157396793365479, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6046317741274834, "clip_ratio/low_mean": 0.12450417224317789, "clip_ratio/low_min": 0.12450417224317789, "clip_ratio/high_mean": 0.09791667014360428, "clip_ratio/high_max": 0.09791667014360428, "clip_ratio/region_mean": 0.22242084238678217, "reward_total_mean": 0.4158652424812317, "reward_meter_mean": 0.2548510432243347, "reward_meter_std": 0.3831472098827362, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9499562978744507, "reward_repeat_soft_std": 0.021399270743131638, "reward_judge_quality_mean": 0.5600000023841858, "reward_judge_quality_std": 0.22258226573467255, "reward_total_composite_mean": 0.4158652424812317, "reward_total_composite_std": 0.10863844305276871} {"timestamp_utc": "2026-04-13T07:33:17Z", "mode": "train", "global_step": 88, "epoch": 0.008839779005524863, "loss": -0.112, "grad_norm": 15.566302299499512, "learning_rate": 9.736363636363637e-06, "num_tokens": 163574.0, "completions/mean_length": 53.25, "completions/min_length": 38.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.25, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.6322734951972961, "rewards/meter/std": 0.4166473150253296, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9894940853118896, "rewards/repeat_soft/std": 0.013136623427271843, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.5233603715896606, "rewards/total_composite/std": 0.11490435898303986, "reward": 0.5233603715896606, "reward_std": 0.11490435898303986, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22387740015983582, "sampling/sampling_logp_difference/max": 1.7036323547363281, "sampling/importance_sampling_ratio/min": 0.1820211559534073, "sampling/importance_sampling_ratio/mean": 1.0229641199111938, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8468092530965805, "clip_ratio/low_mean": 0.09993897192180157, "clip_ratio/low_min": 0.09993897192180157, "clip_ratio/high_mean": 0.1066430639475584, "clip_ratio/high_max": 0.1066430639475584, "clip_ratio/region_mean": 0.20658203586935997, "reward_total_mean": 0.5233603715896606, "reward_meter_mean": 0.6322734951972961, "reward_meter_std": 0.4166473150253296, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9894940853118896, "reward_repeat_soft_std": 0.013136623427271843, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.5233603715896606, "reward_total_composite_std": 0.11490435898303986} {"timestamp_utc": "2026-04-13T07:33:24Z", "mode": "train", "global_step": 89, "epoch": 0.008940231039678554, "loss": 0.052, "grad_norm": 22.00596046447754, "learning_rate": 9.733333333333334e-06, "num_tokens": 165573.0, "completions/mean_length": 84.875, "completions/min_length": 53.0, "completions/max_length": 100.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.875, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 100.0, "rewards/meter/mean": 0.5694794654846191, "rewards/meter/std": 0.32996881008148193, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9958401322364807, "rewards/repeat_soft/std": 0.006230877246707678, "rewards/judge_quality/mean": 0.6449999809265137, "rewards/judge_quality/std": 0.24928471446037292, "rewards/total_composite/mean": 0.5533815622329712, "rewards/total_composite/std": 0.10411175340414047, "reward": 0.5533815622329712, "reward_std": 0.10411174595355988, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23848699033260345, "sampling/sampling_logp_difference/max": 3.618814468383789, "sampling/importance_sampling_ratio/min": 0.026814447715878487, "sampling/importance_sampling_ratio/mean": 0.9860868453979492, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9971832558512688, "clip_ratio/low_mean": 0.07648433186113834, "clip_ratio/low_min": 0.07648433186113834, "clip_ratio/high_mean": 0.09676007181406021, "clip_ratio/high_max": 0.09676007181406021, "clip_ratio/region_mean": 0.17324440367519855, "reward_total_mean": 0.5533815622329712, "reward_meter_mean": 0.5694794654846191, "reward_meter_std": 0.32996881008148193, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9958401322364807, "reward_repeat_soft_std": 0.006230877246707678, "reward_judge_quality_mean": 0.6449999809265137, "reward_judge_quality_std": 0.24928471446037292, "reward_total_composite_mean": 0.5533815622329712, "reward_total_composite_std": 0.10411175340414047} {"timestamp_utc": "2026-04-13T07:33:30Z", "mode": "train", "global_step": 90, "epoch": 0.009040683073832245, "loss": 0.0951, "grad_norm": 23.296001434326172, "learning_rate": 9.730303030303031e-06, "num_tokens": 167021.0, "completions/mean_length": 24.0, "completions/min_length": 21.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.0, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.36750417947769165, "rewards/meter/std": 0.3164251148700714, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9948339462280273, "rewards/repeat_soft/std": 0.01454270351678133, "rewards/judge_quality/mean": 0.6187499761581421, "rewards/judge_quality/std": 0.21866071224212646, "rewards/total_composite/mean": 0.5101072788238525, "rewards/total_composite/std": 0.18115530908107758, "reward": 0.5101072788238525, "reward_std": 0.18115530908107758, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21593332290649414, "sampling/sampling_logp_difference/max": 1.3680660724639893, "sampling/importance_sampling_ratio/min": 0.25459885597229004, "sampling/importance_sampling_ratio/mean": 1.0594345331192017, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3456864058971405, "clip_ratio/low_mean": 0.16109457332640886, "clip_ratio/low_min": 0.16109457332640886, "clip_ratio/high_mean": 0.02095238072797656, "clip_ratio/high_max": 0.02095238072797656, "clip_ratio/region_mean": 0.18204695405438542, "reward_total_mean": 0.5101072788238525, "reward_meter_mean": 0.36750417947769165, "reward_meter_std": 0.3164251148700714, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9948339462280273, "reward_repeat_soft_std": 0.01454270351678133, "reward_judge_quality_mean": 0.6187499761581421, "reward_judge_quality_std": 0.21866071224212646, "reward_total_composite_mean": 0.5101072788238525, "reward_total_composite_std": 0.18115530908107758} {"timestamp_utc": "2026-04-13T07:33:38Z", "mode": "train", "global_step": 91, "epoch": 0.009141135107985936, "loss": -0.0065, "grad_norm": 11.63347053527832, "learning_rate": 9.727272727272728e-06, "num_tokens": 169080.0, "completions/mean_length": 84.375, "completions/min_length": 64.0, "completions/max_length": 106.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.375, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.4681720435619354, "rewards/meter/std": 0.38532817363739014, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.995926022529602, "rewards/repeat_soft/std": 0.003701572772115469, "rewards/judge_quality/mean": 0.5862500071525574, "rewards/judge_quality/std": 0.19190309941768646, "rewards/total_composite/mean": 0.525649905204773, "rewards/total_composite/std": 0.17196428775787354, "reward": 0.525649905204773, "reward_std": 0.17196428775787354, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2211720496416092, "sampling/sampling_logp_difference/max": 1.5955607891082764, "sampling/importance_sampling_ratio/min": 0.20279477536678314, "sampling/importance_sampling_ratio/mean": 1.041763186454773, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2681550085544586, "clip_ratio/low_mean": 0.09809909947216511, "clip_ratio/low_min": 0.09809909947216511, "clip_ratio/high_mean": 0.09866714477539062, "clip_ratio/high_max": 0.09866714477539062, "clip_ratio/region_mean": 0.19676624424755573, "reward_total_mean": 0.525649905204773, "reward_meter_mean": 0.4681720435619354, "reward_meter_std": 0.38532817363739014, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.995926022529602, "reward_repeat_soft_std": 0.003701572772115469, "reward_judge_quality_mean": 0.5862500071525574, "reward_judge_quality_std": 0.19190309941768646, "reward_total_composite_mean": 0.525649905204773, "reward_total_composite_std": 0.17196428775787354} {"timestamp_utc": "2026-04-13T07:33:45Z", "mode": "train", "global_step": 92, "epoch": 0.009241587142139629, "loss": 0.0501, "grad_norm": 17.996219635009766, "learning_rate": 9.724242424242426e-06, "num_tokens": 170756.0, "completions/mean_length": 41.5, "completions/min_length": 32.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.5, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.2540503144264221, "rewards/meter/std": 0.31625130772590637, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9889193773269653, "rewards/repeat_soft/std": 0.01017341110855341, "rewards/judge_quality/mean": 0.6150000095367432, "rewards/judge_quality/std": 0.25286927819252014, "rewards/total_composite/mean": 0.44203996658325195, "rewards/total_composite/std": 0.11635930091142654, "reward": 0.44203996658325195, "reward_std": 0.11635930836200714, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23884731531143188, "sampling/sampling_logp_difference/max": 2.4119834899902344, "sampling/importance_sampling_ratio/min": 0.08963732421398163, "sampling/importance_sampling_ratio/mean": 1.0279514789581299, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8763217777013779, "clip_ratio/low_mean": 0.12165019661188126, "clip_ratio/low_min": 0.12165019661188126, "clip_ratio/high_mean": 0.07047158665955067, "clip_ratio/high_max": 0.07047158665955067, "clip_ratio/region_mean": 0.19212178327143192, "reward_total_mean": 0.44203996658325195, "reward_meter_mean": 0.2540503144264221, "reward_meter_std": 0.31625130772590637, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9889193773269653, "reward_repeat_soft_std": 0.01017341110855341, "reward_judge_quality_mean": 0.6150000095367432, "reward_judge_quality_std": 0.25286927819252014, "reward_total_composite_mean": 0.44203996658325195, "reward_total_composite_std": 0.11635930091142654} {"timestamp_utc": "2026-04-13T07:33:52Z", "mode": "train", "global_step": 93, "epoch": 0.00934203917629332, "loss": -0.004, "grad_norm": 12.886481285095215, "learning_rate": 9.721212121212123e-06, "num_tokens": 172367.0, "completions/mean_length": 54.375, "completions/min_length": 44.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.375, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.667952299118042, "rewards/meter/std": 0.42871978878974915, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9986583590507507, "rewards/repeat_soft/std": 0.0020897281356155872, "rewards/judge_quality/mean": 0.39249998331069946, "rewards/judge_quality/std": 0.08892211318016052, "rewards/total_composite/mean": 0.5257940292358398, "rewards/total_composite/std": 0.11939812451601028, "reward": 0.5257940292358398, "reward_std": 0.11939812451601028, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21974819898605347, "sampling/sampling_logp_difference/max": 2.408405303955078, "sampling/importance_sampling_ratio/min": 0.08995863795280457, "sampling/importance_sampling_ratio/mean": 1.0207183361053467, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.9088242799043655, "clip_ratio/low_mean": 0.09619488101452589, "clip_ratio/low_min": 0.09619488101452589, "clip_ratio/high_mean": 0.11377151682972908, "clip_ratio/high_max": 0.11377151682972908, "clip_ratio/region_mean": 0.20996639784425497, "reward_total_mean": 0.5257940292358398, "reward_meter_mean": 0.667952299118042, "reward_meter_std": 0.42871978878974915, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9986583590507507, "reward_repeat_soft_std": 0.0020897281356155872, "reward_judge_quality_mean": 0.39249998331069946, "reward_judge_quality_std": 0.08892211318016052, "reward_total_composite_mean": 0.5257940292358398, "reward_total_composite_std": 0.11939812451601028} {"timestamp_utc": "2026-04-13T07:33:59Z", "mode": "train", "global_step": 94, "epoch": 0.009442491210447011, "loss": 0.066, "grad_norm": 72.35273742675781, "learning_rate": 9.718181818181818e-06, "num_tokens": 174097.0, "completions/mean_length": 38.25, "completions/min_length": 32.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.25, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.3880566954612732, "rewards/meter/std": 0.3312193751335144, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9880231618881226, "rewards/repeat_soft/std": 0.011508532799780369, "rewards/judge_quality/mean": 0.6812499761581421, "rewards/judge_quality/std": 0.25542333722114563, "rewards/total_composite/mean": 0.5185790061950684, "rewards/total_composite/std": 0.18502680957317352, "reward": 0.5185790061950684, "reward_std": 0.18502680957317352, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20673184096813202, "sampling/sampling_logp_difference/max": 1.6290805339813232, "sampling/importance_sampling_ratio/min": 0.196109801530838, "sampling/importance_sampling_ratio/mean": 1.0550040006637573, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.473860390484333, "clip_ratio/low_mean": 0.11757584288716316, "clip_ratio/low_min": 0.11757584288716316, "clip_ratio/high_mean": 0.061890242621302605, "clip_ratio/high_max": 0.061890242621302605, "clip_ratio/region_mean": 0.17946608550846577, "reward_total_mean": 0.5185790061950684, "reward_meter_mean": 0.3880566954612732, "reward_meter_std": 0.3312193751335144, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9880231618881226, "reward_repeat_soft_std": 0.011508532799780369, "reward_judge_quality_mean": 0.6812499761581421, "reward_judge_quality_std": 0.25542333722114563, "reward_total_composite_mean": 0.5185790061950684, "reward_total_composite_std": 0.18502680957317352} {"timestamp_utc": "2026-04-13T07:34:07Z", "mode": "train", "global_step": 95, "epoch": 0.009542943244600702, "loss": 0.0334, "grad_norm": 10.50815486907959, "learning_rate": 9.715151515151516e-06, "num_tokens": 176393.0, "completions/mean_length": 104.0, "completions/min_length": 75.0, "completions/max_length": 147.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 104.0, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 147.0, "rewards/meter/mean": 0.5511190891265869, "rewards/meter/std": 0.3720203936100006, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9935624003410339, "rewards/repeat_soft/std": 0.009157720021903515, "rewards/judge_quality/mean": 0.5699999928474426, "rewards/judge_quality/std": 0.16035676002502441, "rewards/total_composite/mean": 0.5496877431869507, "rewards/total_composite/std": 0.179593026638031, "reward": 0.5496877431869507, "reward_std": 0.1795930117368698, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23180857300758362, "sampling/sampling_logp_difference/max": 1.9675331115722656, "sampling/importance_sampling_ratio/min": 0.13980130851268768, "sampling/importance_sampling_ratio/mean": 1.0309317111968994, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4507804214954376, "clip_ratio/low_mean": 0.12515215389430523, "clip_ratio/low_min": 0.12515215389430523, "clip_ratio/high_mean": 0.07687505520880222, "clip_ratio/high_max": 0.07687505520880222, "clip_ratio/region_mean": 0.20202720910310745, "reward_total_mean": 0.5496877431869507, "reward_meter_mean": 0.5511190891265869, "reward_meter_std": 0.3720203936100006, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9935624003410339, "reward_repeat_soft_std": 0.009157720021903515, "reward_judge_quality_mean": 0.5699999928474426, "reward_judge_quality_std": 0.16035676002502441, "reward_total_composite_mean": 0.5496877431869507, "reward_total_composite_std": 0.179593026638031} {"timestamp_utc": "2026-04-13T07:34:14Z", "mode": "train", "global_step": 96, "epoch": 0.009643395278754395, "loss": -0.002, "grad_norm": 18.13115692138672, "learning_rate": 9.712121212121213e-06, "num_tokens": 177986.0, "completions/mean_length": 44.125, "completions/min_length": 32.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.125, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.35852187871932983, "rewards/meter/std": 0.3847596347332001, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9928138256072998, "rewards/repeat_soft/std": 0.008599132299423218, "rewards/judge_quality/mean": 0.5575000047683716, "rewards/judge_quality/std": 0.19955308735370636, "rewards/total_composite/mean": 0.45549121499061584, "rewards/total_composite/std": 0.09961428493261337, "reward": 0.45549121499061584, "reward_std": 0.09961428493261337, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21455226838588715, "sampling/sampling_logp_difference/max": 1.6288232803344727, "sampling/importance_sampling_ratio/min": 0.19616025686264038, "sampling/importance_sampling_ratio/mean": 1.0390883684158325, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0448380559682846, "clip_ratio/low_mean": 0.12832126952707767, "clip_ratio/low_min": 0.12832126952707767, "clip_ratio/high_mean": 0.0959168542176485, "clip_ratio/high_max": 0.0959168542176485, "clip_ratio/region_mean": 0.22423812374472618, "reward_total_mean": 0.45549121499061584, "reward_meter_mean": 0.35852187871932983, "reward_meter_std": 0.3847596347332001, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9928138256072998, "reward_repeat_soft_std": 0.008599132299423218, "reward_judge_quality_mean": 0.5575000047683716, "reward_judge_quality_std": 0.19955308735370636, "reward_total_composite_mean": 0.45549121499061584, "reward_total_composite_std": 0.09961428493261337} {"timestamp_utc": "2026-04-13T07:34:22Z", "mode": "train", "global_step": 97, "epoch": 0.009743847312908087, "loss": 0.1103, "grad_norm": 17.4005126953125, "learning_rate": 9.70909090909091e-06, "num_tokens": 179656.0, "completions/mean_length": 43.75, "completions/min_length": 38.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.75, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.4310287833213806, "rewards/meter/std": 0.37155377864837646, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9705358147621155, "rewards/repeat_soft/std": 0.03972465544939041, "rewards/judge_quality/mean": 0.41749998927116394, "rewards/judge_quality/std": 0.06902380287647247, "rewards/total_composite/mean": 0.4592880606651306, "rewards/total_composite/std": 0.10607890784740448, "reward": 0.4592880606651306, "reward_std": 0.10607889294624329, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22239214181900024, "sampling/sampling_logp_difference/max": 1.114335536956787, "sampling/importance_sampling_ratio/min": 0.328133225440979, "sampling/importance_sampling_ratio/mean": 1.0250064134597778, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.259575054049492, "clip_ratio/low_mean": 0.11555228382349014, "clip_ratio/low_min": 0.11555228382349014, "clip_ratio/high_mean": 0.06651191785931587, "clip_ratio/high_max": 0.06651191785931587, "clip_ratio/region_mean": 0.18206420168280602, "reward_total_mean": 0.4592880606651306, "reward_meter_mean": 0.4310287833213806, "reward_meter_std": 0.37155377864837646, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9705358147621155, "reward_repeat_soft_std": 0.03972465544939041, "reward_judge_quality_mean": 0.41749998927116394, "reward_judge_quality_std": 0.06902380287647247, "reward_total_composite_mean": 0.4592880606651306, "reward_total_composite_std": 0.10607890784740448} {"timestamp_utc": "2026-04-13T07:34:30Z", "mode": "train", "global_step": 98, "epoch": 0.009844299347061778, "loss": 0.183, "grad_norm": 18.236595153808594, "learning_rate": 9.706060606060606e-06, "num_tokens": 181422.0, "completions/mean_length": 56.75, "completions/min_length": 42.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.75, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.43153274059295654, "rewards/meter/std": 0.4027426838874817, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9987303614616394, "rewards/repeat_soft/std": 0.0030256821773946285, "rewards/judge_quality/mean": 0.4724999964237213, "rewards/judge_quality/std": 0.19255799055099487, "rewards/total_composite/mean": 0.4896277189254761, "rewards/total_composite/std": 0.12527494132518768, "reward": 0.4896277189254761, "reward_std": 0.12527494132518768, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19202686846256256, "sampling/sampling_logp_difference/max": 1.3069257736206055, "sampling/importance_sampling_ratio/min": 0.27065083384513855, "sampling/importance_sampling_ratio/mean": 1.0375019311904907, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.928257331252098, "clip_ratio/low_mean": 0.08768315427005291, "clip_ratio/low_min": 0.08768315427005291, "clip_ratio/high_mean": 0.08970235660672188, "clip_ratio/high_max": 0.08970235660672188, "clip_ratio/region_mean": 0.1773855108767748, "reward_total_mean": 0.4896277189254761, "reward_meter_mean": 0.43153274059295654, "reward_meter_std": 0.4027426838874817, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9987303614616394, "reward_repeat_soft_std": 0.0030256821773946285, "reward_judge_quality_mean": 0.4724999964237213, "reward_judge_quality_std": 0.19255799055099487, "reward_total_composite_mean": 0.4896277189254761, "reward_total_composite_std": 0.12527494132518768} {"timestamp_utc": "2026-04-13T07:34:36Z", "mode": "train", "global_step": 99, "epoch": 0.009944751381215469, "loss": -0.0123, "grad_norm": 20.899450302124023, "learning_rate": 9.703030303030305e-06, "num_tokens": 182801.0, "completions/mean_length": 28.375, "completions/min_length": 19.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.375, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.24775895476341248, "rewards/meter/std": 0.35025811195373535, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9559658765792847, "rewards/repeat_soft/std": 0.01644347980618477, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.1940544992685318, "rewards/total_composite/mean": 0.3981969356536865, "rewards/total_composite/std": 0.06609706580638885, "reward": 0.3981969356536865, "reward_std": 0.06609705835580826, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17724387347698212, "sampling/sampling_logp_difference/max": 1.2565131187438965, "sampling/importance_sampling_ratio/min": 0.28464481234550476, "sampling/importance_sampling_ratio/mean": 1.0068453550338745, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.540700025856495, "clip_ratio/low_mean": 0.08678907621651888, "clip_ratio/low_min": 0.08678907621651888, "clip_ratio/high_mean": 0.061506545171141624, "clip_ratio/high_max": 0.061506545171141624, "clip_ratio/region_mean": 0.1482956213876605, "reward_total_mean": 0.3981969356536865, "reward_meter_mean": 0.24775895476341248, "reward_meter_std": 0.35025811195373535, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9559658765792847, "reward_repeat_soft_std": 0.01644347980618477, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.1940544992685318, "reward_total_composite_mean": 0.3981969356536865, "reward_total_composite_std": 0.06609706580638885} {"timestamp_utc": "2026-04-13T07:34:44Z", "mode": "train", "global_step": 100, "epoch": 0.010045203415369162, "loss": 0.1025, "grad_norm": 15.105440139770508, "learning_rate": 9.7e-06, "num_tokens": 184736.0, "completions/mean_length": 67.875, "completions/min_length": 54.0, "completions/max_length": 92.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.875, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 92.0, "rewards/meter/mean": 0.5348031520843506, "rewards/meter/std": 0.3817266523838043, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9921558499336243, "rewards/repeat_soft/std": 0.0050133876502513885, "rewards/judge_quality/mean": 0.5024999976158142, "rewards/judge_quality/std": 0.1440981924533844, "rewards/total_composite/mean": 0.4639267325401306, "rewards/total_composite/std": 0.20549696683883667, "reward": 0.4639267325401306, "reward_std": 0.20549695193767548, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21573399007320404, "sampling/sampling_logp_difference/max": 1.7488975524902344, "sampling/importance_sampling_ratio/min": 0.17396561801433563, "sampling/importance_sampling_ratio/mean": 1.0013856887817383, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9593983590602875, "clip_ratio/low_mean": 0.06655851006507874, "clip_ratio/low_min": 0.06655851006507874, "clip_ratio/high_mean": 0.13855982199311256, "clip_ratio/high_max": 0.13855982199311256, "clip_ratio/region_mean": 0.2051183320581913, "reward_total_mean": 0.4639267325401306, "reward_meter_mean": 0.5348031520843506, "reward_meter_std": 0.3817266523838043, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9921558499336243, "reward_repeat_soft_std": 0.0050133876502513885, "reward_judge_quality_mean": 0.5024999976158142, "reward_judge_quality_std": 0.1440981924533844, "reward_total_composite_mean": 0.4639267325401306, "reward_total_composite_std": 0.20549696683883667} {"timestamp_utc": "2026-04-13T07:35:34Z", "mode": "eval", "global_step": 100, "epoch": 0.010045203415369162, "eval_loss": NaN, "eval_runtime": 50.6952, "eval_samples_per_second": 1.578, "eval_steps_per_second": 0.197, "eval_num_tokens": 184736.0, "eval_completions/mean_length": 79.6375, "eval_completions/min_length": 32.0, "eval_completions/max_length": 143.1, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 79.6375, "eval_completions/min_terminated_length": 32.0, "eval_completions/max_terminated_length": 143.1, "eval_rewards/meter/mean": 0.4947461932897568, "eval_rewards/meter/std": 0.3523126423358917, "eval_rewards/count_adherence/mean": 0.9681249976158142, "eval_rewards/count_adherence/std": 0.09015611484646797, "eval_rewards/hard_gate/mean": 0.9375, "eval_rewards/hard_gate/std": 0.1767766922712326, "eval_rewards/repeat_soft/mean": 0.991607666015625, "eval_rewards/repeat_soft/std": 0.011974084354005755, "eval_rewards/judge_quality/mean": 0.47200000286102295, "eval_rewards/judge_quality/std": 0.1661804124712944, "eval_rewards/total_composite/mean": 0.471698522567749, "eval_rewards/total_composite/std": 0.17879592925310134, "eval_reward": 0.471698522567749, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.14445659518241882, "eval_sampling/sampling_logp_difference/max": 1.225595474243164, "eval_sampling/importance_sampling_ratio/min": 0.2977974995970726, "eval_sampling/importance_sampling_ratio/mean": 1.0436278223991393, "eval_sampling/importance_sampling_ratio/max": 1.5806872248649597, "eval_entropy": 2.184915065765381, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.471698522567749, "eval_reward_meter_mean": 0.4947461932897568, "eval_reward_meter_std": 0.3523126423358917, "eval_reward_count_adherence_mean": 0.9681249976158142, "eval_reward_count_adherence_std": 0.09015611484646797, "eval_reward_hard_gate_mean": 0.9375, "eval_reward_hard_gate_std": 0.1767766922712326, "eval_reward_repeat_soft_mean": 0.991607666015625, "eval_reward_repeat_soft_std": 0.011974084354005755, "eval_reward_judge_quality_mean": 0.47200000286102295, "eval_reward_judge_quality_std": 0.1661804124712944, "eval_reward_total_composite_mean": 0.471698522567749, "eval_reward_total_composite_std": 0.17879592925310134} {"timestamp_utc": "2026-04-13T07:35:45Z", "mode": "train", "global_step": 101, "epoch": 0.010145655449522853, "loss": 0.0118, "grad_norm": 16.414012908935547, "learning_rate": 9.696969696969698e-06, "num_tokens": 186614.0, "completions/mean_length": 58.75, "completions/min_length": 45.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.75, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.31307607889175415, "rewards/meter/std": 0.37023577094078064, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9960094690322876, "rewards/repeat_soft/std": 0.003800690406933427, "rewards/judge_quality/mean": 0.5737500190734863, "rewards/judge_quality/std": 0.1566559225320816, "rewards/total_composite/mean": 0.41067010164260864, "rewards/total_composite/std": 0.2034657597541809, "reward": 0.41067010164260864, "reward_std": 0.2034657597541809, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23996204137802124, "sampling/sampling_logp_difference/max": 1.938732624053955, "sampling/importance_sampling_ratio/min": 0.17366401851177216, "sampling/importance_sampling_ratio/mean": 1.0507534742355347, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.916314333677292, "clip_ratio/low_mean": 0.09495464991778135, "clip_ratio/low_min": 0.09495464991778135, "clip_ratio/high_mean": 0.1007721908390522, "clip_ratio/high_max": 0.1007721908390522, "clip_ratio/region_mean": 0.19572684075683355, "reward_total_mean": 0.41067010164260864, "reward_meter_mean": 0.31307607889175415, "reward_meter_std": 0.37023577094078064, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9960094690322876, "reward_repeat_soft_std": 0.003800690406933427, "reward_judge_quality_mean": 0.5737500190734863, "reward_judge_quality_std": 0.1566559225320816, "reward_total_composite_mean": 0.41067010164260864, "reward_total_composite_std": 0.2034657597541809} {"timestamp_utc": "2026-04-13T07:35:53Z", "mode": "train", "global_step": 102, "epoch": 0.010246107483676544, "loss": 0.0199, "grad_norm": 13.145352363586426, "learning_rate": 9.693939393939395e-06, "num_tokens": 188740.0, "completions/mean_length": 86.75, "completions/min_length": 79.0, "completions/max_length": 111.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.75, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.32552945613861084, "rewards/meter/std": 0.29995009303092957, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9968774914741516, "rewards/repeat_soft/std": 0.002367391949519515, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.43840116262435913, "rewards/total_composite/std": 0.0820188969373703, "reward": 0.43840116262435913, "reward_std": 0.08201887458562851, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21868892014026642, "sampling/sampling_logp_difference/max": 2.1058950424194336, "sampling/importance_sampling_ratio/min": 0.12173666059970856, "sampling/importance_sampling_ratio/mean": 1.0365042686462402, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.044892191886902, "clip_ratio/low_mean": 0.11399291269481182, "clip_ratio/low_min": 0.11399291269481182, "clip_ratio/high_mean": 0.07613622397184372, "clip_ratio/high_max": 0.07613622397184372, "clip_ratio/region_mean": 0.19012913666665554, "reward_total_mean": 0.43840116262435913, "reward_meter_mean": 0.32552945613861084, "reward_meter_std": 0.29995009303092957, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9968774914741516, "reward_repeat_soft_std": 0.002367391949519515, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.43840116262435913, "reward_total_composite_std": 0.0820188969373703} {"timestamp_utc": "2026-04-13T07:36:01Z", "mode": "train", "global_step": 103, "epoch": 0.010346559517830235, "loss": 0.0483, "grad_norm": 13.322657585144043, "learning_rate": 9.690909090909092e-06, "num_tokens": 190428.0, "completions/mean_length": 54.0, "completions/min_length": 41.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.4907858967781067, "rewards/meter/std": 0.4426969289779663, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9974976778030396, "rewards/repeat_soft/std": 0.0037490674294531345, "rewards/judge_quality/mean": 0.5562499761581421, "rewards/judge_quality/std": 0.2249404788017273, "rewards/total_composite/mean": 0.5294002294540405, "rewards/total_composite/std": 0.14981630444526672, "reward": 0.5294002294540405, "reward_std": 0.14981631934642792, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1901864856481552, "sampling/sampling_logp_difference/max": 1.7494490146636963, "sampling/importance_sampling_ratio/min": 0.17386971414089203, "sampling/importance_sampling_ratio/mean": 1.0129625797271729, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3739588409662247, "clip_ratio/low_mean": 0.06334241945296526, "clip_ratio/low_min": 0.06334241945296526, "clip_ratio/high_mean": 0.09503825195133686, "clip_ratio/high_max": 0.09503825195133686, "clip_ratio/region_mean": 0.15838067140430212, "reward_total_mean": 0.5294002294540405, "reward_meter_mean": 0.4907858967781067, "reward_meter_std": 0.4426969289779663, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9974976778030396, "reward_repeat_soft_std": 0.0037490674294531345, "reward_judge_quality_mean": 0.5562499761581421, "reward_judge_quality_std": 0.2249404788017273, "reward_total_composite_mean": 0.5294002294540405, "reward_total_composite_std": 0.14981630444526672} {"timestamp_utc": "2026-04-13T07:36:08Z", "mode": "train", "global_step": 104, "epoch": 0.010447011551983928, "loss": 0.067, "grad_norm": 15.824663162231445, "learning_rate": 9.687878787878788e-06, "num_tokens": 192042.0, "completions/mean_length": 47.75, "completions/min_length": 41.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.75, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.4273160696029663, "rewards/meter/std": 0.3913634121417999, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9948374032974243, "rewards/repeat_soft/std": 0.011940538883209229, "rewards/judge_quality/mean": 0.6525000333786011, "rewards/judge_quality/std": 0.2418234497308731, "rewards/total_composite/mean": 0.5368877649307251, "rewards/total_composite/std": 0.18454858660697937, "reward": 0.5368877649307251, "reward_std": 0.18454860150814056, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2321016639471054, "sampling/sampling_logp_difference/max": 2.423257827758789, "sampling/importance_sampling_ratio/min": 0.08863239735364914, "sampling/importance_sampling_ratio/mean": 1.0372896194458008, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.88126140832901, "clip_ratio/low_mean": 0.10984764993190765, "clip_ratio/low_min": 0.10984764993190765, "clip_ratio/high_mean": 0.08070892840623856, "clip_ratio/high_max": 0.08070892840623856, "clip_ratio/region_mean": 0.1905565783381462, "reward_total_mean": 0.5368877649307251, "reward_meter_mean": 0.4273160696029663, "reward_meter_std": 0.3913634121417999, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9948374032974243, "reward_repeat_soft_std": 0.011940538883209229, "reward_judge_quality_mean": 0.6525000333786011, "reward_judge_quality_std": 0.2418234497308731, "reward_total_composite_mean": 0.5368877649307251, "reward_total_composite_std": 0.18454858660697937} {"timestamp_utc": "2026-04-13T07:36:15Z", "mode": "train", "global_step": 105, "epoch": 0.01054746358613762, "loss": -0.0929, "grad_norm": 10.761244773864746, "learning_rate": 9.684848484848487e-06, "num_tokens": 194281.0, "completions/mean_length": 95.875, "completions/min_length": 68.0, "completions/max_length": 123.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.875, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.3006367087364197, "rewards/meter/std": 0.3225688934326172, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9966142773628235, "rewards/repeat_soft/std": 0.004816299770027399, "rewards/judge_quality/mean": 0.41874998807907104, "rewards/judge_quality/std": 0.14574319124221802, "rewards/total_composite/mean": 0.45476049184799194, "rewards/total_composite/std": 0.15194669365882874, "reward": 0.45476049184799194, "reward_std": 0.15194669365882874, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2330923229455948, "sampling/sampling_logp_difference/max": 1.461348533630371, "sampling/importance_sampling_ratio/min": 0.23192331194877625, "sampling/importance_sampling_ratio/mean": 1.0213702917099, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.5699221789836884, "clip_ratio/low_mean": 0.17626286298036575, "clip_ratio/low_min": 0.17626286298036575, "clip_ratio/high_mean": 0.048399388790130615, "clip_ratio/high_max": 0.048399388790130615, "clip_ratio/region_mean": 0.22466225177049637, "reward_total_mean": 0.45476049184799194, "reward_meter_mean": 0.3006367087364197, "reward_meter_std": 0.3225688934326172, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9966142773628235, "reward_repeat_soft_std": 0.004816299770027399, "reward_judge_quality_mean": 0.41874998807907104, "reward_judge_quality_std": 0.14574319124221802, "reward_total_composite_mean": 0.45476049184799194, "reward_total_composite_std": 0.15194669365882874} {"timestamp_utc": "2026-04-13T07:36:23Z", "mode": "train", "global_step": 106, "epoch": 0.01064791562029131, "loss": 0.045, "grad_norm": 32.41508102416992, "learning_rate": 9.681818181818182e-06, "num_tokens": 195731.0, "completions/mean_length": 29.25, "completions/min_length": 26.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.25, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.41725194454193115, "rewards/meter/std": 0.2775874435901642, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9898457527160645, "rewards/repeat_soft/std": 0.005739502143114805, "rewards/judge_quality/mean": 0.5275000333786011, "rewards/judge_quality/std": 0.1954299360513687, "rewards/total_composite/mean": 0.5062490105628967, "rewards/total_composite/std": 0.13873079419136047, "reward": 0.5062490105628967, "reward_std": 0.13873079419136047, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19110196828842163, "sampling/sampling_logp_difference/max": 1.9213972091674805, "sampling/importance_sampling_ratio/min": 0.1464022547006607, "sampling/importance_sampling_ratio/mean": 1.0243908166885376, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9342062026262283, "clip_ratio/low_mean": 0.1261307019740343, "clip_ratio/low_min": 0.1261307019740343, "clip_ratio/high_mean": 0.04126082360744476, "clip_ratio/high_max": 0.04126082360744476, "clip_ratio/region_mean": 0.16739152558147907, "reward_total_mean": 0.5062490105628967, "reward_meter_mean": 0.41725194454193115, "reward_meter_std": 0.2775874435901642, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9898457527160645, "reward_repeat_soft_std": 0.005739502143114805, "reward_judge_quality_mean": 0.5275000333786011, "reward_judge_quality_std": 0.1954299360513687, "reward_total_composite_mean": 0.5062490105628967, "reward_total_composite_std": 0.13873079419136047} {"timestamp_utc": "2026-04-13T07:36:30Z", "mode": "train", "global_step": 107, "epoch": 0.010748367654445002, "loss": 0.2057, "grad_norm": 20.082307815551758, "learning_rate": 9.67878787878788e-06, "num_tokens": 197443.0, "completions/mean_length": 51.0, "completions/min_length": 31.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.0, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.7254915833473206, "rewards/meter/std": 0.40892869234085083, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9917322993278503, "rewards/repeat_soft/std": 0.010131089948117733, "rewards/judge_quality/mean": 0.7362500429153442, "rewards/judge_quality/std": 0.25376805663108826, "rewards/total_composite/mean": 0.7275059223175049, "rewards/total_composite/std": 0.27270805835723877, "reward": 0.7275059223175049, "reward_std": 0.27270805835723877, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17575828731060028, "sampling/sampling_logp_difference/max": 1.2130756378173828, "sampling/importance_sampling_ratio/min": 0.29728156328201294, "sampling/importance_sampling_ratio/mean": 1.0152204036712646, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3037724867463112, "clip_ratio/low_mean": 0.04798919567838311, "clip_ratio/low_min": 0.04798919567838311, "clip_ratio/high_mean": 0.12696912698447704, "clip_ratio/high_max": 0.12696912698447704, "clip_ratio/region_mean": 0.17495832266286016, "reward_total_mean": 0.7275059223175049, "reward_meter_mean": 0.7254915833473206, "reward_meter_std": 0.40892869234085083, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9917322993278503, "reward_repeat_soft_std": 0.010131089948117733, "reward_judge_quality_mean": 0.7362500429153442, "reward_judge_quality_std": 0.25376805663108826, "reward_total_composite_mean": 0.7275059223175049, "reward_total_composite_std": 0.27270805835723877} {"timestamp_utc": "2026-04-13T07:36:38Z", "mode": "train", "global_step": 108, "epoch": 0.010848819688598695, "loss": -0.0728, "grad_norm": 31.107200622558594, "learning_rate": 9.675757575757577e-06, "num_tokens": 198743.0, "completions/mean_length": 27.5, "completions/min_length": 21.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.5, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.17419485747814178, "rewards/meter/std": 0.17303234338760376, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9976261854171753, "rewards/repeat_soft/std": 0.003726511262357235, "rewards/judge_quality/mean": 0.5350000262260437, "rewards/judge_quality/std": 0.1911618709564209, "rewards/total_composite/mean": 0.40953513979911804, "rewards/total_composite/std": 0.061801183968782425, "reward": 0.40953513979911804, "reward_std": 0.061801180243492126, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.31676381826400757, "sampling/sampling_logp_difference/max": 2.9225621223449707, "sampling/importance_sampling_ratio/min": 0.05379568040370941, "sampling/importance_sampling_ratio/mean": 1.0342049598693848, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0506207197904587, "clip_ratio/low_mean": 0.15618032403290272, "clip_ratio/low_min": 0.15618032403290272, "clip_ratio/high_mean": 0.10346023179590702, "clip_ratio/high_max": 0.10346023179590702, "clip_ratio/region_mean": 0.25964055582880974, "reward_total_mean": 0.40953513979911804, "reward_meter_mean": 0.17419485747814178, "reward_meter_std": 0.17303234338760376, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9976261854171753, "reward_repeat_soft_std": 0.003726511262357235, "reward_judge_quality_mean": 0.5350000262260437, "reward_judge_quality_std": 0.1911618709564209, "reward_total_composite_mean": 0.40953513979911804, "reward_total_composite_std": 0.061801183968782425} {"timestamp_utc": "2026-04-13T07:36:45Z", "mode": "train", "global_step": 109, "epoch": 0.010949271722752386, "loss": 0.0143, "grad_norm": 13.207579612731934, "learning_rate": 9.672727272727274e-06, "num_tokens": 200420.0, "completions/mean_length": 49.625, "completions/min_length": 33.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.625, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9199807643890381, "rewards/meter/std": 0.18390671908855438, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9963053464889526, "rewards/repeat_soft/std": 0.004458059091120958, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.5862847566604614, "rewards/total_composite/std": 0.26184654235839844, "reward": 0.5862847566604614, "reward_std": 0.26184654235839844, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22561202943325043, "sampling/sampling_logp_difference/max": 1.3455278873443604, "sampling/importance_sampling_ratio/min": 0.26040223240852356, "sampling/importance_sampling_ratio/mean": 1.032443881034851, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4901363253593445, "clip_ratio/low_mean": 0.02556818164885044, "clip_ratio/low_min": 0.02556818164885044, "clip_ratio/high_mean": 0.20468470267951488, "clip_ratio/high_max": 0.20468470267951488, "clip_ratio/region_mean": 0.23025288432836533, "reward_total_mean": 0.5862847566604614, "reward_meter_mean": 0.9199807643890381, "reward_meter_std": 0.18390671908855438, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9963053464889526, "reward_repeat_soft_std": 0.004458059091120958, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.5862847566604614, "reward_total_composite_std": 0.26184654235839844} {"timestamp_utc": "2026-04-13T07:36:51Z", "mode": "train", "global_step": 110, "epoch": 0.011049723756906077, "loss": 0.0721, "grad_norm": 13.896294593811035, "learning_rate": 9.66969696969697e-06, "num_tokens": 202156.0, "completions/mean_length": 48.0, "completions/min_length": 37.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.0, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.892151951789856, "rewards/meter/std": 0.26865240931510925, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9856277704238892, "rewards/repeat_soft/std": 0.017225177958607674, "rewards/judge_quality/mean": 0.5974999666213989, "rewards/judge_quality/std": 0.220891073346138, "rewards/total_composite/mean": 0.6859848499298096, "rewards/total_composite/std": 0.16468684375286102, "reward": 0.6859848499298096, "reward_std": 0.16468684375286102, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2160102128982544, "sampling/sampling_logp_difference/max": 1.8444128036499023, "sampling/importance_sampling_ratio/min": 0.1581181436777115, "sampling/importance_sampling_ratio/mean": 1.036145806312561, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3849086612462997, "clip_ratio/low_mean": 0.11906185001134872, "clip_ratio/low_min": 0.11906185001134872, "clip_ratio/high_mean": 0.048634571954607964, "clip_ratio/high_max": 0.048634571954607964, "clip_ratio/region_mean": 0.1676964219659567, "reward_total_mean": 0.6859848499298096, "reward_meter_mean": 0.892151951789856, "reward_meter_std": 0.26865240931510925, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9856277704238892, "reward_repeat_soft_std": 0.017225177958607674, "reward_judge_quality_mean": 0.5974999666213989, "reward_judge_quality_std": 0.220891073346138, "reward_total_composite_mean": 0.6859848499298096, "reward_total_composite_std": 0.16468684375286102} {"timestamp_utc": "2026-04-13T07:36:59Z", "mode": "train", "global_step": 111, "epoch": 0.01115017579105977, "loss": -0.0285, "grad_norm": 10.08155632019043, "learning_rate": 9.666666666666667e-06, "num_tokens": 204200.0, "completions/mean_length": 91.5, "completions/min_length": 69.0, "completions/max_length": 127.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.5, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 127.0, "rewards/meter/mean": 0.6964003443717957, "rewards/meter/std": 0.3417958915233612, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9986571073532104, "rewards/repeat_soft/std": 0.0013369194930419326, "rewards/judge_quality/mean": 0.5112500190734863, "rewards/judge_quality/std": 0.18216457962989807, "rewards/total_composite/mean": 0.5833487510681152, "rewards/total_composite/std": 0.14542630314826965, "reward": 0.5833487510681152, "reward_std": 0.14542630314826965, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22398188710212708, "sampling/sampling_logp_difference/max": 1.491401195526123, "sampling/importance_sampling_ratio/min": 0.22505709528923035, "sampling/importance_sampling_ratio/mean": 1.0558103322982788, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.0628055930137634, "clip_ratio/low_mean": 0.09790320880711079, "clip_ratio/low_min": 0.09790320880711079, "clip_ratio/high_mean": 0.12199797295033932, "clip_ratio/high_max": 0.12199797295033932, "clip_ratio/region_mean": 0.2199011817574501, "reward_total_mean": 0.5833487510681152, "reward_meter_mean": 0.6964003443717957, "reward_meter_std": 0.3417958915233612, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9986571073532104, "reward_repeat_soft_std": 0.0013369194930419326, "reward_judge_quality_mean": 0.5112500190734863, "reward_judge_quality_std": 0.18216457962989807, "reward_total_composite_mean": 0.5833487510681152, "reward_total_composite_std": 0.14542630314826965} {"timestamp_utc": "2026-04-13T07:37:06Z", "mode": "train", "global_step": 112, "epoch": 0.011250627825213461, "loss": 0.1087, "grad_norm": 20.6037540435791, "learning_rate": 9.663636363636364e-06, "num_tokens": 205731.0, "completions/mean_length": 26.375, "completions/min_length": 16.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.375, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.5949082970619202, "rewards/meter/std": 0.494957834482193, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9459680318832397, "rewards/repeat_soft/std": 0.04101009666919708, "rewards/judge_quality/mean": 0.5600000023841858, "rewards/judge_quality/std": 0.22258225083351135, "rewards/total_composite/mean": 0.5486740469932556, "rewards/total_composite/std": 0.20209264755249023, "reward": 0.5486740469932556, "reward_std": 0.20209261775016785, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18613271415233612, "sampling/sampling_logp_difference/max": 1.6638908386230469, "sampling/importance_sampling_ratio/min": 0.18940061330795288, "sampling/importance_sampling_ratio/mean": 1.0181546211242676, "sampling/importance_sampling_ratio/max": 1.8430614471435547, "entropy": 1.4548956751823425, "clip_ratio/low_mean": 0.06781775318086147, "clip_ratio/low_min": 0.06781775318086147, "clip_ratio/high_mean": 0.10600329004228115, "clip_ratio/high_max": 0.10600329004228115, "clip_ratio/region_mean": 0.17382104322314262, "reward_total_mean": 0.5486740469932556, "reward_meter_mean": 0.5949082970619202, "reward_meter_std": 0.494957834482193, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9459680318832397, "reward_repeat_soft_std": 0.04101009666919708, "reward_judge_quality_mean": 0.5600000023841858, "reward_judge_quality_std": 0.22258225083351135, "reward_total_composite_mean": 0.5486740469932556, "reward_total_composite_std": 0.20209264755249023} {"timestamp_utc": "2026-04-13T07:37:13Z", "mode": "train", "global_step": 113, "epoch": 0.011351079859367152, "loss": 0.0605, "grad_norm": 11.709718704223633, "learning_rate": 9.660606060606061e-06, "num_tokens": 207592.0, "completions/mean_length": 76.625, "completions/min_length": 64.0, "completions/max_length": 92.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.625, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 92.0, "rewards/meter/mean": 0.5275776386260986, "rewards/meter/std": 0.32127389311790466, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9927095770835876, "rewards/repeat_soft/std": 0.007404064293950796, "rewards/judge_quality/mean": 0.5824999809265137, "rewards/judge_quality/std": 0.23260943591594696, "rewards/total_composite/mean": 0.5425221920013428, "rewards/total_composite/std": 0.16789241135120392, "reward": 0.5425221920013428, "reward_std": 0.16789241135120392, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2142861932516098, "sampling/sampling_logp_difference/max": 2.0963268280029297, "sampling/importance_sampling_ratio/min": 0.12290704995393753, "sampling/importance_sampling_ratio/mean": 1.0310522317886353, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.829342469573021, "clip_ratio/low_mean": 0.09125383384525776, "clip_ratio/low_min": 0.09125383384525776, "clip_ratio/high_mean": 0.07866137474775314, "clip_ratio/high_max": 0.07866137474775314, "clip_ratio/region_mean": 0.1699152085930109, "reward_total_mean": 0.5425221920013428, "reward_meter_mean": 0.5275776386260986, "reward_meter_std": 0.32127389311790466, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9927095770835876, "reward_repeat_soft_std": 0.007404064293950796, "reward_judge_quality_mean": 0.5824999809265137, "reward_judge_quality_std": 0.23260943591594696, "reward_total_composite_mean": 0.5425221920013428, "reward_total_composite_std": 0.16789241135120392} {"timestamp_utc": "2026-04-13T07:37:20Z", "mode": "train", "global_step": 114, "epoch": 0.011451531893520843, "loss": -0.0008, "grad_norm": 15.91455078125, "learning_rate": 9.657575757575758e-06, "num_tokens": 209166.0, "completions/mean_length": 42.75, "completions/min_length": 32.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.75, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.41170966625213623, "rewards/meter/std": 0.39287254214286804, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9963371753692627, "rewards/repeat_soft/std": 0.0069928220473229885, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.3950405418872833, "rewards/total_composite/std": 0.20517849922180176, "reward": 0.3950405418872833, "reward_std": 0.20517848432064056, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23370611667633057, "sampling/sampling_logp_difference/max": 1.5829181671142578, "sampling/importance_sampling_ratio/min": 0.20537489652633667, "sampling/importance_sampling_ratio/mean": 1.0398354530334473, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.164140671491623, "clip_ratio/low_mean": 0.11847196333110332, "clip_ratio/low_min": 0.11847196333110332, "clip_ratio/high_mean": 0.08506148867309093, "clip_ratio/high_max": 0.08506148867309093, "clip_ratio/region_mean": 0.20353345200419426, "reward_total_mean": 0.3950405418872833, "reward_meter_mean": 0.41170966625213623, "reward_meter_std": 0.39287254214286804, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9963371753692627, "reward_repeat_soft_std": 0.0069928220473229885, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.3950405418872833, "reward_total_composite_std": 0.20517849922180176} {"timestamp_utc": "2026-04-13T07:37:28Z", "mode": "train", "global_step": 115, "epoch": 0.011551983927674536, "loss": -0.1652, "grad_norm": 10.372628211975098, "learning_rate": 9.654545454545456e-06, "num_tokens": 211540.0, "completions/mean_length": 108.75, "completions/min_length": 62.0, "completions/max_length": 148.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.75, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 148.0, "rewards/meter/mean": 0.3633410930633545, "rewards/meter/std": 0.29584023356437683, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.997858464717865, "rewards/repeat_soft/std": 0.0016820140881463885, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.21357084810733795, "rewards/total_composite/mean": 0.41038256883621216, "rewards/total_composite/std": 0.1801348179578781, "reward": 0.41038256883621216, "reward_std": 0.1801348179578781, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2164894938468933, "sampling/sampling_logp_difference/max": 1.768528938293457, "sampling/importance_sampling_ratio/min": 0.17058373987674713, "sampling/importance_sampling_ratio/mean": 1.0498042106628418, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.7530274242162704, "clip_ratio/low_mean": 0.051655348390340805, "clip_ratio/low_min": 0.051655348390340805, "clip_ratio/high_mean": 0.16114997677505016, "clip_ratio/high_max": 0.16114997677505016, "clip_ratio/region_mean": 0.21280532516539097, "reward_total_mean": 0.41038256883621216, "reward_meter_mean": 0.3633410930633545, "reward_meter_std": 0.29584023356437683, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.997858464717865, "reward_repeat_soft_std": 0.0016820140881463885, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.21357084810733795, "reward_total_composite_mean": 0.41038256883621216, "reward_total_composite_std": 0.1801348179578781} {"timestamp_utc": "2026-04-13T07:37:35Z", "mode": "train", "global_step": 116, "epoch": 0.011652435961828227, "loss": 0.0623, "grad_norm": 15.952447891235352, "learning_rate": 9.651515151515153e-06, "num_tokens": 213129.0, "completions/mean_length": 43.625, "completions/min_length": 30.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.625, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.8196585178375244, "rewards/meter/std": 0.2894671559333801, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9935790300369263, "rewards/repeat_soft/std": 0.007324355188757181, "rewards/judge_quality/mean": 0.5199999809265137, "rewards/judge_quality/std": 0.19272483885288239, "rewards/total_composite/mean": 0.615305483341217, "rewards/total_composite/std": 0.15007352828979492, "reward": 0.615305483341217, "reward_std": 0.15007352828979492, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2312517613172531, "sampling/sampling_logp_difference/max": 2.045539140701294, "sampling/importance_sampling_ratio/min": 0.12931044399738312, "sampling/importance_sampling_ratio/mean": 1.0233547687530518, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1754041761159897, "clip_ratio/low_mean": 0.08337220642715693, "clip_ratio/low_min": 0.08337220642715693, "clip_ratio/high_mean": 0.12287236005067825, "clip_ratio/high_max": 0.12287236005067825, "clip_ratio/region_mean": 0.20624456647783518, "reward_total_mean": 0.615305483341217, "reward_meter_mean": 0.8196585178375244, "reward_meter_std": 0.2894671559333801, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9935790300369263, "reward_repeat_soft_std": 0.007324355188757181, "reward_judge_quality_mean": 0.5199999809265137, "reward_judge_quality_std": 0.19272483885288239, "reward_total_composite_mean": 0.615305483341217, "reward_total_composite_std": 0.15007352828979492} {"timestamp_utc": "2026-04-13T07:37:42Z", "mode": "train", "global_step": 117, "epoch": 0.011752887995981919, "loss": 0.0196, "grad_norm": 11.291123390197754, "learning_rate": 9.648484848484849e-06, "num_tokens": 214683.0, "completions/mean_length": 43.25, "completions/min_length": 38.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.25, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9655974507331848, "rewards/meter/std": 0.01152831595391035, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.989479660987854, "rewards/repeat_soft/std": 0.006363891065120697, "rewards/judge_quality/mean": 0.9099999666213989, "rewards/judge_quality/std": 0.018516412004828453, "rewards/total_composite/mean": 0.9195917248725891, "rewards/total_composite/std": 0.014142941683530807, "reward": 0.9195917248725891, "reward_std": 0.014142945408821106, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07092545926570892, "sampling/sampling_logp_difference/max": 1.4284789562225342, "sampling/importance_sampling_ratio/min": 0.3051346242427826, "sampling/importance_sampling_ratio/mean": 1.006468653678894, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.338104959577322, "clip_ratio/low_mean": 0.012290635146200657, "clip_ratio/low_min": 0.012290635146200657, "clip_ratio/high_mean": 0.02475078939460218, "clip_ratio/high_max": 0.02475078939460218, "clip_ratio/region_mean": 0.037041424540802836, "reward_total_mean": 0.9195917248725891, "reward_meter_mean": 0.9655974507331848, "reward_meter_std": 0.01152831595391035, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.989479660987854, "reward_repeat_soft_std": 0.006363891065120697, "reward_judge_quality_mean": 0.9099999666213989, "reward_judge_quality_std": 0.018516412004828453, "reward_total_composite_mean": 0.9195917248725891, "reward_total_composite_std": 0.014142941683530807} {"timestamp_utc": "2026-04-13T07:37:50Z", "mode": "train", "global_step": 118, "epoch": 0.01185334003013561, "loss": -0.0076, "grad_norm": 7.859068393707275, "learning_rate": 9.645454545454548e-06, "num_tokens": 217324.0, "completions/mean_length": 130.125, "completions/min_length": 86.0, "completions/max_length": 184.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 130.125, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 184.0, "rewards/meter/mean": 0.5821857452392578, "rewards/meter/std": 0.3516339659690857, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9981836080551147, "rewards/repeat_soft/std": 0.0015714566688984632, "rewards/judge_quality/mean": 0.42250001430511475, "rewards/judge_quality/std": 0.14616528153419495, "rewards/total_composite/mean": 0.3869349956512451, "rewards/total_composite/std": 0.2658952474594116, "reward": 0.3869349956512451, "reward_std": 0.2658952474594116, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23853881657123566, "sampling/sampling_logp_difference/max": 1.860891342163086, "sampling/importance_sampling_ratio/min": 0.15553393959999084, "sampling/importance_sampling_ratio/mean": 1.0505948066711426, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.004988729953766, "clip_ratio/low_mean": 0.03663211967796087, "clip_ratio/low_min": 0.03663211967796087, "clip_ratio/high_mean": 0.15321064181625843, "clip_ratio/high_max": 0.15321064181625843, "clip_ratio/region_mean": 0.1898427614942193, "reward_total_mean": 0.3869349956512451, "reward_meter_mean": 0.5821857452392578, "reward_meter_std": 0.3516339659690857, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9981836080551147, "reward_repeat_soft_std": 0.0015714566688984632, "reward_judge_quality_mean": 0.42250001430511475, "reward_judge_quality_std": 0.14616528153419495, "reward_total_composite_mean": 0.3869349956512451, "reward_total_composite_std": 0.2658952474594116} {"timestamp_utc": "2026-04-13T07:37:58Z", "mode": "train", "global_step": 119, "epoch": 0.011953792064289303, "loss": 0.2217, "grad_norm": 18.535816192626953, "learning_rate": 9.642424242424243e-06, "num_tokens": 219114.0, "completions/mean_length": 57.75, "completions/min_length": 35.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.75, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.403515487909317, "rewards/meter/std": 0.4413924217224121, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9988487362861633, "rewards/repeat_soft/std": 0.0016559003852307796, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.12772038578987122, "rewards/total_composite/mean": 0.4148494601249695, "rewards/total_composite/std": 0.202467143535614, "reward": 0.4148494601249695, "reward_std": 0.20246712863445282, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21171781420707703, "sampling/sampling_logp_difference/max": 1.7705450057983398, "sampling/importance_sampling_ratio/min": 0.17024017870426178, "sampling/importance_sampling_ratio/mean": 1.0579570531845093, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.507441684603691, "clip_ratio/low_mean": 0.09634115733206272, "clip_ratio/low_min": 0.09634115733206272, "clip_ratio/high_mean": 0.07929547782987356, "clip_ratio/high_max": 0.07929547782987356, "clip_ratio/region_mean": 0.17563663516193628, "reward_total_mean": 0.4148494601249695, "reward_meter_mean": 0.403515487909317, "reward_meter_std": 0.4413924217224121, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9988487362861633, "reward_repeat_soft_std": 0.0016559003852307796, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.12772038578987122, "reward_total_composite_mean": 0.4148494601249695, "reward_total_composite_std": 0.202467143535614} {"timestamp_utc": "2026-04-13T07:38:05Z", "mode": "train", "global_step": 120, "epoch": 0.012054244098442994, "loss": 0.0885, "grad_norm": 11.159661293029785, "learning_rate": 9.63939393939394e-06, "num_tokens": 221565.0, "completions/mean_length": 112.375, "completions/min_length": 63.0, "completions/max_length": 125.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.375, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.4886738955974579, "rewards/meter/std": 0.4274240732192993, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9982452988624573, "rewards/repeat_soft/std": 0.0026251429226249456, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.4151157736778259, "rewards/total_composite/std": 0.20270900428295135, "reward": 0.4151157736778259, "reward_std": 0.20270900428295135, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2313220500946045, "sampling/sampling_logp_difference/max": 1.854095458984375, "sampling/importance_sampling_ratio/min": 0.15659452974796295, "sampling/importance_sampling_ratio/mean": 1.014096736907959, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4326031506061554, "clip_ratio/low_mean": 0.11750435456633568, "clip_ratio/low_min": 0.11750435456633568, "clip_ratio/high_mean": 0.1027077529579401, "clip_ratio/high_max": 0.1027077529579401, "clip_ratio/region_mean": 0.22021210752427578, "reward_total_mean": 0.4151157736778259, "reward_meter_mean": 0.4886738955974579, "reward_meter_std": 0.4274240732192993, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9982452988624573, "reward_repeat_soft_std": 0.0026251429226249456, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.4151157736778259, "reward_total_composite_std": 0.20270900428295135} {"timestamp_utc": "2026-04-13T07:38:13Z", "mode": "train", "global_step": 121, "epoch": 0.012154696132596685, "loss": 0.0284, "grad_norm": 9.35704231262207, "learning_rate": 9.636363636363638e-06, "num_tokens": 224368.0, "completions/mean_length": 150.375, "completions/min_length": 101.0, "completions/max_length": 173.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 150.375, "completions/min_terminated_length": 101.0, "completions/max_terminated_length": 173.0, "rewards/meter/mean": 0.30038923025131226, "rewards/meter/std": 0.26126354932785034, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9967457056045532, "rewards/repeat_soft/std": 0.0029344093054533005, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4315181374549866, "rewards/total_composite/std": 0.07135103642940521, "reward": 0.4315181374549866, "reward_std": 0.07135103642940521, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21734307706356049, "sampling/sampling_logp_difference/max": 1.5603547096252441, "sampling/importance_sampling_ratio/min": 0.21006155014038086, "sampling/importance_sampling_ratio/mean": 1.047584891319275, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4637745916843414, "clip_ratio/low_mean": 0.10287107527256012, "clip_ratio/low_min": 0.10287107527256012, "clip_ratio/high_mean": 0.11393089592456818, "clip_ratio/high_max": 0.11393089592456818, "clip_ratio/region_mean": 0.2168019711971283, "reward_total_mean": 0.4315181374549866, "reward_meter_mean": 0.30038923025131226, "reward_meter_std": 0.26126354932785034, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9967457056045532, "reward_repeat_soft_std": 0.0029344093054533005, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4315181374549866, "reward_total_composite_std": 0.07135103642940521} {"timestamp_utc": "2026-04-13T07:38:21Z", "mode": "train", "global_step": 122, "epoch": 0.012255148166750376, "loss": 0.4034, "grad_norm": 11.443964004516602, "learning_rate": 9.633333333333335e-06, "num_tokens": 226239.0, "completions/mean_length": 75.875, "completions/min_length": 38.0, "completions/max_length": 175.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.875, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 175.0, "rewards/meter/mean": 0.24370630085468292, "rewards/meter/std": 0.3257603347301483, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.3720119297504425, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9960834383964539, "rewards/repeat_soft/std": 0.005711476318538189, "rewards/judge_quality/mean": 0.6775000095367432, "rewards/judge_quality/std": 0.16446886956691742, "rewards/total_composite/mean": 0.4244243800640106, "rewards/total_composite/std": 0.19604572653770447, "reward": 0.4244243800640106, "reward_std": 0.19604569673538208, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20754988491535187, "sampling/sampling_logp_difference/max": 1.750192642211914, "sampling/importance_sampling_ratio/min": 0.1737404763698578, "sampling/importance_sampling_ratio/mean": 1.04361891746521, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.415245831012726, "clip_ratio/low_mean": 0.09463843237608671, "clip_ratio/low_min": 0.09463843237608671, "clip_ratio/high_mean": 0.07074670866131783, "clip_ratio/high_max": 0.07074670866131783, "clip_ratio/region_mean": 0.16538514103740454, "reward_total_mean": 0.4244243800640106, "reward_meter_mean": 0.24370630085468292, "reward_meter_std": 0.3257603347301483, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.3720119297504425, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9960834383964539, "reward_repeat_soft_std": 0.005711476318538189, "reward_judge_quality_mean": 0.6775000095367432, "reward_judge_quality_std": 0.16446886956691742, "reward_total_composite_mean": 0.4244243800640106, "reward_total_composite_std": 0.19604572653770447} {"timestamp_utc": "2026-04-13T07:38:28Z", "mode": "train", "global_step": 123, "epoch": 0.012355600200904069, "loss": 0.0613, "grad_norm": 18.49869155883789, "learning_rate": 9.63030303030303e-06, "num_tokens": 227650.0, "completions/mean_length": 28.375, "completions/min_length": 24.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.375, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.6262902021408081, "rewards/meter/std": 0.3038237392902374, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9953283071517944, "rewards/repeat_soft/std": 0.0074143316596746445, "rewards/judge_quality/mean": 0.6525000333786011, "rewards/judge_quality/std": 0.2921227812767029, "rewards/total_composite/mean": 0.6145520210266113, "rewards/total_composite/std": 0.19700059294700623, "reward": 0.6145520210266113, "reward_std": 0.19700059294700623, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21750284731388092, "sampling/sampling_logp_difference/max": 2.3387670516967773, "sampling/importance_sampling_ratio/min": 0.09644648432731628, "sampling/importance_sampling_ratio/mean": 0.9930812120437622, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5528540685772896, "clip_ratio/low_mean": 0.14069444499909878, "clip_ratio/low_min": 0.14069444499909878, "clip_ratio/high_mean": 0.08239214681088924, "clip_ratio/high_max": 0.08239214681088924, "clip_ratio/region_mean": 0.22308659180998802, "reward_total_mean": 0.6145520210266113, "reward_meter_mean": 0.6262902021408081, "reward_meter_std": 0.3038237392902374, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9953283071517944, "reward_repeat_soft_std": 0.0074143316596746445, "reward_judge_quality_mean": 0.6525000333786011, "reward_judge_quality_std": 0.2921227812767029, "reward_total_composite_mean": 0.6145520210266113, "reward_total_composite_std": 0.19700059294700623} {"timestamp_utc": "2026-04-13T07:38:34Z", "mode": "train", "global_step": 124, "epoch": 0.01245605223505776, "loss": 0.2265, "grad_norm": 35.97916030883789, "learning_rate": 9.627272727272728e-06, "num_tokens": 229010.0, "completions/mean_length": 21.0, "completions/min_length": 14.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.0, "completions/min_terminated_length": 14.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.8410463333129883, "rewards/meter/std": 0.26125726103782654, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9526292085647583, "rewards/repeat_soft/std": 0.03851768746972084, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.549912691116333, "rewards/total_composite/std": 0.12555654346942902, "reward": 0.549912691116333, "reward_std": 0.12555654346942902, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16508203744888306, "sampling/sampling_logp_difference/max": 0.9596343040466309, "sampling/importance_sampling_ratio/min": 0.419406920671463, "sampling/importance_sampling_ratio/mean": 1.040798306465149, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4324931278824806, "clip_ratio/low_mean": 0.04584703966975212, "clip_ratio/low_min": 0.04584703966975212, "clip_ratio/high_mean": 0.09017446916550398, "clip_ratio/high_max": 0.09017446916550398, "clip_ratio/region_mean": 0.1360215088352561, "reward_total_mean": 0.549912691116333, "reward_meter_mean": 0.8410463333129883, "reward_meter_std": 0.26125726103782654, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9526292085647583, "reward_repeat_soft_std": 0.03851768746972084, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.549912691116333, "reward_total_composite_std": 0.12555654346942902} {"timestamp_utc": "2026-04-13T07:38:42Z", "mode": "train", "global_step": 125, "epoch": 0.012556504269211451, "loss": 0.0966, "grad_norm": 17.960975646972656, "learning_rate": 9.624242424242425e-06, "num_tokens": 230885.0, "completions/mean_length": 65.375, "completions/min_length": 52.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.375, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.31092584133148193, "rewards/meter/std": 0.3357381820678711, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9908581972122192, "rewards/repeat_soft/std": 0.009348123334348202, "rewards/judge_quality/mean": 0.5325000286102295, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.45793652534484863, "rewards/total_composite/std": 0.1227300614118576, "reward": 0.45793652534484863, "reward_std": 0.1227300614118576, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.27371904253959656, "sampling/sampling_logp_difference/max": 2.2968673706054688, "sampling/importance_sampling_ratio/min": 0.10057341307401657, "sampling/importance_sampling_ratio/mean": 0.9990410804748535, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9411287307739258, "clip_ratio/low_mean": 0.1552787609398365, "clip_ratio/low_min": 0.1552787609398365, "clip_ratio/high_mean": 0.09186454489827156, "clip_ratio/high_max": 0.09186454489827156, "clip_ratio/region_mean": 0.24714330583810806, "reward_total_mean": 0.45793652534484863, "reward_meter_mean": 0.31092584133148193, "reward_meter_std": 0.3357381820678711, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9908581972122192, "reward_repeat_soft_std": 0.009348123334348202, "reward_judge_quality_mean": 0.5325000286102295, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.45793652534484863, "reward_total_composite_std": 0.1227300614118576} {"timestamp_utc": "2026-04-13T07:38:51Z", "mode": "train", "global_step": 126, "epoch": 0.012656956303365142, "loss": 0.134, "grad_norm": 24.076974868774414, "learning_rate": 9.621212121212122e-06, "num_tokens": 232246.0, "completions/mean_length": 26.125, "completions/min_length": 19.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.125, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.7440218925476074, "rewards/meter/std": 0.45724499225616455, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9560876488685608, "rewards/repeat_soft/std": 0.018136821687221527, "rewards/judge_quality/mean": 0.38875001668930054, "rewards/judge_quality/std": 0.25709572434425354, "rewards/total_composite/mean": 0.508181095123291, "rewards/total_composite/std": 0.13012929260730743, "reward": 0.508181095123291, "reward_std": 0.13012929260730743, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21434858441352844, "sampling/sampling_logp_difference/max": 1.1329364776611328, "sampling/importance_sampling_ratio/min": 0.32208606600761414, "sampling/importance_sampling_ratio/mean": 1.0196175575256348, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0628183633089066, "clip_ratio/low_mean": 0.0923573998734355, "clip_ratio/low_min": 0.0923573998734355, "clip_ratio/high_mean": 0.06937899254262447, "clip_ratio/high_max": 0.06937899254262447, "clip_ratio/region_mean": 0.16173639241605997, "reward_total_mean": 0.508181095123291, "reward_meter_mean": 0.7440218925476074, "reward_meter_std": 0.45724499225616455, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9560876488685608, "reward_repeat_soft_std": 0.018136821687221527, "reward_judge_quality_mean": 0.38875001668930054, "reward_judge_quality_std": 0.25709572434425354, "reward_total_composite_mean": 0.508181095123291, "reward_total_composite_std": 0.13012929260730743} {"timestamp_utc": "2026-04-13T07:38:58Z", "mode": "train", "global_step": 127, "epoch": 0.012757408337518835, "loss": 0.1658, "grad_norm": 15.085627555847168, "learning_rate": 9.61818181818182e-06, "num_tokens": 233806.0, "completions/mean_length": 49.0, "completions/min_length": 36.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.0, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.5365213751792908, "rewards/meter/std": 0.4541638195514679, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.993897557258606, "rewards/repeat_soft/std": 0.009357492439448833, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.17136012017726898, "rewards/total_composite/mean": 0.4945714473724365, "rewards/total_composite/std": 0.14548209309577942, "reward": 0.4945714473724365, "reward_std": 0.14548209309577942, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21035538613796234, "sampling/sampling_logp_difference/max": 1.697453498840332, "sampling/importance_sampling_ratio/min": 0.1831493079662323, "sampling/importance_sampling_ratio/mean": 1.0626816749572754, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.212236076593399, "clip_ratio/low_mean": 0.07253067381680012, "clip_ratio/low_min": 0.07253067381680012, "clip_ratio/high_mean": 0.10060209594666958, "clip_ratio/high_max": 0.10060209594666958, "clip_ratio/region_mean": 0.1731327697634697, "reward_total_mean": 0.4945714473724365, "reward_meter_mean": 0.5365213751792908, "reward_meter_std": 0.4541638195514679, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.993897557258606, "reward_repeat_soft_std": 0.009357492439448833, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.17136012017726898, "reward_total_composite_mean": 0.4945714473724365, "reward_total_composite_std": 0.14548209309577942} {"timestamp_utc": "2026-04-13T07:39:06Z", "mode": "train", "global_step": 128, "epoch": 0.012857860371672527, "loss": -0.0173, "grad_norm": 11.304403305053711, "learning_rate": 9.615151515151517e-06, "num_tokens": 235866.0, "completions/mean_length": 84.5, "completions/min_length": 61.0, "completions/max_length": 110.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.5, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 110.0, "rewards/meter/mean": 0.42456066608428955, "rewards/meter/std": 0.3047213554382324, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.993170440196991, "rewards/repeat_soft/std": 0.009959491901099682, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.21357084810733795, "rewards/total_composite/mean": 0.4069315791130066, "rewards/total_composite/std": 0.17672067880630493, "reward": 0.4069315791130066, "reward_std": 0.17672067880630493, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2357020229101181, "sampling/sampling_logp_difference/max": 1.8493890762329102, "sampling/importance_sampling_ratio/min": 0.15733325481414795, "sampling/importance_sampling_ratio/mean": 1.0373672246932983, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6177062541246414, "clip_ratio/low_mean": 0.05069653131067753, "clip_ratio/low_min": 0.05069653131067753, "clip_ratio/high_mean": 0.17823937721550465, "clip_ratio/high_max": 0.17823937721550465, "clip_ratio/region_mean": 0.22893590852618217, "reward_total_mean": 0.4069315791130066, "reward_meter_mean": 0.42456066608428955, "reward_meter_std": 0.3047213554382324, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.993170440196991, "reward_repeat_soft_std": 0.009959491901099682, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.21357084810733795, "reward_total_composite_mean": 0.4069315791130066, "reward_total_composite_std": 0.17672067880630493} {"timestamp_utc": "2026-04-13T07:39:14Z", "mode": "train", "global_step": 129, "epoch": 0.012958312405826218, "loss": 0.071, "grad_norm": 11.652441024780273, "learning_rate": 9.612121212121212e-06, "num_tokens": 237693.0, "completions/mean_length": 75.375, "completions/min_length": 55.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.375, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.3495359718799591, "rewards/meter/std": 0.26254764199256897, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9941217303276062, "rewards/repeat_soft/std": 0.014426924288272858, "rewards/judge_quality/mean": 0.4362499713897705, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.40984025597572327, "rewards/total_composite/std": 0.19771966338157654, "reward": 0.40984025597572327, "reward_std": 0.19771966338157654, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23216703534126282, "sampling/sampling_logp_difference/max": 1.5401573181152344, "sampling/importance_sampling_ratio/min": 0.21434737741947174, "sampling/importance_sampling_ratio/mean": 1.0318115949630737, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.789889633655548, "clip_ratio/low_mean": 0.10355722904205322, "clip_ratio/low_min": 0.10355722904205322, "clip_ratio/high_mean": 0.10569983534514904, "clip_ratio/high_max": 0.10569983534514904, "clip_ratio/region_mean": 0.20925706438720226, "reward_total_mean": 0.40984025597572327, "reward_meter_mean": 0.3495359718799591, "reward_meter_std": 0.26254764199256897, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9941217303276062, "reward_repeat_soft_std": 0.014426924288272858, "reward_judge_quality_mean": 0.4362499713897705, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.40984025597572327, "reward_total_composite_std": 0.19771966338157654} {"timestamp_utc": "2026-04-13T07:39:23Z", "mode": "train", "global_step": 130, "epoch": 0.013058764439979909, "loss": 0.0814, "grad_norm": 8.03388500213623, "learning_rate": 9.60909090909091e-06, "num_tokens": 240318.0, "completions/mean_length": 151.125, "completions/min_length": 109.0, "completions/max_length": 209.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 151.125, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 209.0, "rewards/meter/mean": 0.4799237847328186, "rewards/meter/std": 0.3395111560821533, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.13363061845302582, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9976893663406372, "rewards/repeat_soft/std": 0.00220813718624413, "rewards/judge_quality/mean": 0.2887499928474426, "rewards/judge_quality/std": 0.11630470305681229, "rewards/total_composite/mean": 0.42769867181777954, "rewards/total_composite/std": 0.09508629143238068, "reward": 0.42769867181777954, "reward_std": 0.09508628398180008, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2239220142364502, "sampling/sampling_logp_difference/max": 1.2956266403198242, "sampling/importance_sampling_ratio/min": 0.2811238467693329, "sampling/importance_sampling_ratio/mean": 1.0429434776306152, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.850192368030548, "clip_ratio/low_mean": 0.09250323288142681, "clip_ratio/low_min": 0.09250323288142681, "clip_ratio/high_mean": 0.10934280045330524, "clip_ratio/high_max": 0.10934280045330524, "clip_ratio/region_mean": 0.20184603333473206, "reward_total_mean": 0.42769867181777954, "reward_meter_mean": 0.4799237847328186, "reward_meter_std": 0.3395111560821533, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.13363061845302582, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9976893663406372, "reward_repeat_soft_std": 0.00220813718624413, "reward_judge_quality_mean": 0.2887499928474426, "reward_judge_quality_std": 0.11630470305681229, "reward_total_composite_mean": 0.42769867181777954, "reward_total_composite_std": 0.09508629143238068} {"timestamp_utc": "2026-04-13T07:39:33Z", "mode": "train", "global_step": 131, "epoch": 0.013159216474133602, "loss": 0.0504, "grad_norm": 23.772933959960938, "learning_rate": 9.606060606060607e-06, "num_tokens": 241857.0, "completions/mean_length": 34.375, "completions/min_length": 26.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.375, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.5936883091926575, "rewards/meter/std": 0.38848671317100525, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9913143515586853, "rewards/repeat_soft/std": 0.009590408764779568, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.5737825632095337, "rewards/total_composite/std": 0.18015919625759125, "reward": 0.5737825632095337, "reward_std": 0.18015921115875244, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2230602502822876, "sampling/sampling_logp_difference/max": 1.8174285888671875, "sampling/importance_sampling_ratio/min": 0.16244292259216309, "sampling/importance_sampling_ratio/mean": 1.016609787940979, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.539472445845604, "clip_ratio/low_mean": 0.0953866122290492, "clip_ratio/low_min": 0.0953866122290492, "clip_ratio/high_mean": 0.13186106085777283, "clip_ratio/high_max": 0.13186106085777283, "clip_ratio/region_mean": 0.22724767308682203, "reward_total_mean": 0.5737825632095337, "reward_meter_mean": 0.5936883091926575, "reward_meter_std": 0.38848671317100525, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9913143515586853, "reward_repeat_soft_std": 0.009590408764779568, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.5737825632095337, "reward_total_composite_std": 0.18015919625759125} {"timestamp_utc": "2026-04-13T07:39:41Z", "mode": "train", "global_step": 132, "epoch": 0.013259668508287293, "loss": 0.105, "grad_norm": 10.050190925598145, "learning_rate": 9.603030303030304e-06, "num_tokens": 244452.0, "completions/mean_length": 121.375, "completions/min_length": 90.0, "completions/max_length": 171.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 121.375, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 171.0, "rewards/meter/mean": 0.4083891212940216, "rewards/meter/std": 0.3109019994735718, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.1035098284482956, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9966987371444702, "rewards/repeat_soft/std": 0.0018282209057360888, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.445995032787323, "rewards/total_composite/std": 0.09745653718709946, "reward": 0.445995032787323, "reward_std": 0.09745653718709946, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21541813015937805, "sampling/sampling_logp_difference/max": 1.6822853088378906, "sampling/importance_sampling_ratio/min": 0.18594853579998016, "sampling/importance_sampling_ratio/mean": 1.0369393825531006, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3659809827804565, "clip_ratio/low_mean": 0.09735992550849915, "clip_ratio/low_min": 0.09735992550849915, "clip_ratio/high_mean": 0.1149887666106224, "clip_ratio/high_max": 0.1149887666106224, "clip_ratio/region_mean": 0.21234869211912155, "reward_total_mean": 0.445995032787323, "reward_meter_mean": 0.4083891212940216, "reward_meter_std": 0.3109019994735718, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.1035098284482956, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9966987371444702, "reward_repeat_soft_std": 0.0018282209057360888, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.445995032787323, "reward_total_composite_std": 0.09745653718709946} {"timestamp_utc": "2026-04-13T07:39:49Z", "mode": "train", "global_step": 133, "epoch": 0.013360120542440984, "loss": 0.1965, "grad_norm": 11.466776847839355, "learning_rate": 9.600000000000001e-06, "num_tokens": 246683.0, "completions/mean_length": 79.875, "completions/min_length": 45.0, "completions/max_length": 136.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.875, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 136.0, "rewards/meter/mean": 0.5764611959457397, "rewards/meter/std": 0.38859090209007263, "rewards/count_adherence/mean": 0.8333333730697632, "rewards/count_adherence/std": 0.17817415297031403, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.995432436466217, "rewards/repeat_soft/std": 0.006603735499083996, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.5122246742248535, "rewards/total_composite/std": 0.2017168402671814, "reward": 0.5122246742248535, "reward_std": 0.2017168253660202, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2326447069644928, "sampling/sampling_logp_difference/max": 2.0507612228393555, "sampling/importance_sampling_ratio/min": 0.12863695621490479, "sampling/importance_sampling_ratio/mean": 1.0416052341461182, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.39143967628479, "clip_ratio/low_mean": 0.11269427556544542, "clip_ratio/low_min": 0.11269427556544542, "clip_ratio/high_mean": 0.09823915176093578, "clip_ratio/high_max": 0.09823915176093578, "clip_ratio/region_mean": 0.2109334273263812, "reward_total_mean": 0.5122246742248535, "reward_meter_mean": 0.5764611959457397, "reward_meter_std": 0.38859090209007263, "reward_count_adherence_mean": 0.8333333730697632, "reward_count_adherence_std": 0.17817415297031403, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.995432436466217, "reward_repeat_soft_std": 0.006603735499083996, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.5122246742248535, "reward_total_composite_std": 0.2017168402671814} {"timestamp_utc": "2026-04-13T07:39:57Z", "mode": "train", "global_step": 134, "epoch": 0.013460572576594675, "loss": 0.1005, "grad_norm": 9.484580993652344, "learning_rate": 9.596969696969699e-06, "num_tokens": 249180.0, "completions/mean_length": 119.125, "completions/min_length": 94.0, "completions/max_length": 146.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 119.125, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 146.0, "rewards/meter/mean": 0.889968991279602, "rewards/meter/std": 0.2062651365995407, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9984543919563293, "rewards/repeat_soft/std": 0.001125063979998231, "rewards/judge_quality/mean": 0.5112500190734863, "rewards/judge_quality/std": 0.18216457962989807, "rewards/total_composite/mean": 0.6544889807701111, "rewards/total_composite/std": 0.12818054854869843, "reward": 0.6544889807701111, "reward_std": 0.12818053364753723, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24180176854133606, "sampling/sampling_logp_difference/max": 1.5038063526153564, "sampling/importance_sampling_ratio/min": 0.22228246927261353, "sampling/importance_sampling_ratio/mean": 1.0537770986557007, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.092375487089157, "clip_ratio/low_mean": 0.12049133330583572, "clip_ratio/low_min": 0.12049133330583572, "clip_ratio/high_mean": 0.08003133162856102, "clip_ratio/high_max": 0.08003133162856102, "clip_ratio/region_mean": 0.20052266493439674, "reward_total_mean": 0.6544889807701111, "reward_meter_mean": 0.889968991279602, "reward_meter_std": 0.2062651365995407, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9984543919563293, "reward_repeat_soft_std": 0.001125063979998231, "reward_judge_quality_mean": 0.5112500190734863, "reward_judge_quality_std": 0.18216457962989807, "reward_total_composite_mean": 0.6544889807701111, "reward_total_composite_std": 0.12818054854869843} {"timestamp_utc": "2026-04-13T07:40:06Z", "mode": "train", "global_step": 135, "epoch": 0.013561024610748368, "loss": 0.0019, "grad_norm": 9.059500694274902, "learning_rate": 9.593939393939394e-06, "num_tokens": 251497.0, "completions/mean_length": 123.625, "completions/min_length": 87.0, "completions/max_length": 165.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 123.625, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 165.0, "rewards/meter/mean": 0.3144863247871399, "rewards/meter/std": 0.28785303235054016, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.13363061845302582, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9965611100196838, "rewards/repeat_soft/std": 0.003245075698941946, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.4088372588157654, "rewards/total_composite/std": 0.08868332952260971, "reward": 0.4088372588157654, "reward_std": 0.08868333697319031, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22240547835826874, "sampling/sampling_logp_difference/max": 1.379908561706543, "sampling/importance_sampling_ratio/min": 0.25160157680511475, "sampling/importance_sampling_ratio/mean": 1.0371609926223755, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.719191998243332, "clip_ratio/low_mean": 0.1372978687286377, "clip_ratio/low_min": 0.1372978687286377, "clip_ratio/high_mean": 0.05146290548145771, "clip_ratio/high_max": 0.05146290548145771, "clip_ratio/region_mean": 0.1887607742100954, "reward_total_mean": 0.4088372588157654, "reward_meter_mean": 0.3144863247871399, "reward_meter_std": 0.28785303235054016, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.13363061845302582, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9965611100196838, "reward_repeat_soft_std": 0.003245075698941946, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.4088372588157654, "reward_total_composite_std": 0.08868332952260971} {"timestamp_utc": "2026-04-13T07:40:13Z", "mode": "train", "global_step": 136, "epoch": 0.01366147664490206, "loss": 0.0238, "grad_norm": 11.203330993652344, "learning_rate": 9.590909090909091e-06, "num_tokens": 253602.0, "completions/mean_length": 83.125, "completions/min_length": 70.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 83.125, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.4576031267642975, "rewards/meter/std": 0.43045464158058167, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.996076226234436, "rewards/repeat_soft/std": 0.004953309893608093, "rewards/judge_quality/mean": 0.6325000524520874, "rewards/judge_quality/std": 0.18850921094417572, "rewards/total_composite/mean": 0.5137057304382324, "rewards/total_composite/std": 0.15345917642116547, "reward": 0.5137057304382324, "reward_std": 0.15345917642116547, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20802229642868042, "sampling/sampling_logp_difference/max": 1.8483209609985352, "sampling/importance_sampling_ratio/min": 0.15750139951705933, "sampling/importance_sampling_ratio/mean": 1.0360591411590576, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.956837147474289, "clip_ratio/low_mean": 0.10638447385281324, "clip_ratio/low_min": 0.10638447385281324, "clip_ratio/high_mean": 0.06874127313494682, "clip_ratio/high_max": 0.06874127313494682, "clip_ratio/region_mean": 0.17512574698776007, "reward_total_mean": 0.5137057304382324, "reward_meter_mean": 0.4576031267642975, "reward_meter_std": 0.43045464158058167, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.996076226234436, "reward_repeat_soft_std": 0.004953309893608093, "reward_judge_quality_mean": 0.6325000524520874, "reward_judge_quality_std": 0.18850921094417572, "reward_total_composite_mean": 0.5137057304382324, "reward_total_composite_std": 0.15345917642116547} {"timestamp_utc": "2026-04-13T07:40:19Z", "mode": "train", "global_step": 137, "epoch": 0.01376192867905575, "loss": -0.0257, "grad_norm": 18.757707595825195, "learning_rate": 9.587878787878789e-06, "num_tokens": 255110.0, "completions/mean_length": 34.5, "completions/min_length": 30.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.5, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.42969757318496704, "rewards/meter/std": 0.3924821615219116, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.997260332107544, "rewards/repeat_soft/std": 0.004701441153883934, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.5069931745529175, "rewards/total_composite/std": 0.19512850046157837, "reward": 0.5069931745529175, "reward_std": 0.19512848556041718, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19561952352523804, "sampling/sampling_logp_difference/max": 1.0618122816085815, "sampling/importance_sampling_ratio/min": 0.34582850337028503, "sampling/importance_sampling_ratio/mean": 1.0478719472885132, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3865096867084503, "clip_ratio/low_mean": 0.11821861192584038, "clip_ratio/low_min": 0.11821861192584038, "clip_ratio/high_mean": 0.08298234827816486, "clip_ratio/high_max": 0.08298234827816486, "clip_ratio/region_mean": 0.20120096020400524, "reward_total_mean": 0.5069931745529175, "reward_meter_mean": 0.42969757318496704, "reward_meter_std": 0.3924821615219116, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.997260332107544, "reward_repeat_soft_std": 0.004701441153883934, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.5069931745529175, "reward_total_composite_std": 0.19512850046157837} {"timestamp_utc": "2026-04-13T07:40:26Z", "mode": "train", "global_step": 138, "epoch": 0.013862380713209442, "loss": 0.2324, "grad_norm": 19.762535095214844, "learning_rate": 9.584848484848486e-06, "num_tokens": 256848.0, "completions/mean_length": 48.25, "completions/min_length": 33.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.25, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.22157180309295654, "rewards/meter/std": 0.23181016743183136, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9917092323303223, "rewards/repeat_soft/std": 0.012141245417296886, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.42576566338539124, "rewards/total_composite/std": 0.1713506579399109, "reward": 0.42576566338539124, "reward_std": 0.17135067284107208, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.27887052297592163, "sampling/sampling_logp_difference/max": 2.26450514793396, "sampling/importance_sampling_ratio/min": 0.10388142615556717, "sampling/importance_sampling_ratio/mean": 1.038446307182312, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.365349918603897, "clip_ratio/low_mean": 0.13626768812537193, "clip_ratio/low_min": 0.13626768812537193, "clip_ratio/high_mean": 0.0834330152720213, "clip_ratio/high_max": 0.0834330152720213, "clip_ratio/region_mean": 0.21970070339739323, "reward_total_mean": 0.42576566338539124, "reward_meter_mean": 0.22157180309295654, "reward_meter_std": 0.23181016743183136, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9917092323303223, "reward_repeat_soft_std": 0.012141245417296886, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.42576566338539124, "reward_total_composite_std": 0.1713506579399109} {"timestamp_utc": "2026-04-13T07:40:33Z", "mode": "train", "global_step": 139, "epoch": 0.013962832747363135, "loss": 0.0099, "grad_norm": 15.954649925231934, "learning_rate": 9.581818181818181e-06, "num_tokens": 258427.0, "completions/mean_length": 46.375, "completions/min_length": 37.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.375, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.4192202091217041, "rewards/meter/std": 0.4678597152233124, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9996633529663086, "rewards/repeat_soft/std": 0.0009521605097688735, "rewards/judge_quality/mean": 0.4937500059604645, "rewards/judge_quality/std": 0.1728696972131729, "rewards/total_composite/mean": 0.46494266390800476, "rewards/total_composite/std": 0.14755646884441376, "reward": 0.46494266390800476, "reward_std": 0.14755645394325256, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2324858456850052, "sampling/sampling_logp_difference/max": 1.674997329711914, "sampling/importance_sampling_ratio/min": 0.1873086839914322, "sampling/importance_sampling_ratio/mean": 1.0463426113128662, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6474010050296783, "clip_ratio/low_mean": 0.10105736553668976, "clip_ratio/low_min": 0.10105736553668976, "clip_ratio/high_mean": 0.11296234279870987, "clip_ratio/high_max": 0.11296234279870987, "clip_ratio/region_mean": 0.21401970833539963, "reward_total_mean": 0.46494266390800476, "reward_meter_mean": 0.4192202091217041, "reward_meter_std": 0.4678597152233124, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9996633529663086, "reward_repeat_soft_std": 0.0009521605097688735, "reward_judge_quality_mean": 0.4937500059604645, "reward_judge_quality_std": 0.1728696972131729, "reward_total_composite_mean": 0.46494266390800476, "reward_total_composite_std": 0.14755646884441376} {"timestamp_utc": "2026-04-13T07:40:41Z", "mode": "train", "global_step": 140, "epoch": 0.014063284781516826, "loss": 0.1831, "grad_norm": 20.739917755126953, "learning_rate": 9.57878787878788e-06, "num_tokens": 260043.0, "completions/mean_length": 44.0, "completions/min_length": 23.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.0, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.4250248372554779, "rewards/meter/std": 0.34178006649017334, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9861264824867249, "rewards/repeat_soft/std": 0.023440638557076454, "rewards/judge_quality/mean": 0.44875001907348633, "rewards/judge_quality/std": 0.2125651240348816, "rewards/total_composite/mean": 0.4422690272331238, "rewards/total_composite/std": 0.09150385111570358, "reward": 0.4422690272331238, "reward_std": 0.09150385111570358, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23629844188690186, "sampling/sampling_logp_difference/max": 1.9918994903564453, "sampling/importance_sampling_ratio/min": 0.13643601536750793, "sampling/importance_sampling_ratio/mean": 0.9938191771507263, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3093688562512398, "clip_ratio/low_mean": 0.11244106199592352, "clip_ratio/low_min": 0.11244106199592352, "clip_ratio/high_mean": 0.09376941062510014, "clip_ratio/high_max": 0.09376941062510014, "clip_ratio/region_mean": 0.20621047262102365, "reward_total_mean": 0.4422690272331238, "reward_meter_mean": 0.4250248372554779, "reward_meter_std": 0.34178006649017334, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9861264824867249, "reward_repeat_soft_std": 0.023440638557076454, "reward_judge_quality_mean": 0.44875001907348633, "reward_judge_quality_std": 0.2125651240348816, "reward_total_composite_mean": 0.4422690272331238, "reward_total_composite_std": 0.09150385111570358} {"timestamp_utc": "2026-04-13T07:40:49Z", "mode": "train", "global_step": 141, "epoch": 0.014163736815670517, "loss": -0.0717, "grad_norm": 7.565834999084473, "learning_rate": 9.575757575757576e-06, "num_tokens": 263212.0, "completions/mean_length": 190.125, "completions/min_length": 153.0, "completions/max_length": 233.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 190.125, "completions/min_terminated_length": 153.0, "completions/max_terminated_length": 233.0, "rewards/meter/mean": 0.45693451166152954, "rewards/meter/std": 0.2026921510696411, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9970571994781494, "rewards/repeat_soft/std": 0.002340377075597644, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.10925068706274033, "rewards/total_composite/mean": 0.35164132714271545, "rewards/total_composite/std": 0.22201529145240784, "reward": 0.35164132714271545, "reward_std": 0.22201529145240784, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21230649948120117, "sampling/sampling_logp_difference/max": 1.8322811126708984, "sampling/importance_sampling_ratio/min": 0.16004806756973267, "sampling/importance_sampling_ratio/mean": 1.0359517335891724, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3753039687871933, "clip_ratio/low_mean": 0.057767802849411964, "clip_ratio/low_min": 0.057767802849411964, "clip_ratio/high_mean": 0.15601530484855175, "clip_ratio/high_max": 0.15601530484855175, "clip_ratio/region_mean": 0.21378310769796371, "reward_total_mean": 0.35164132714271545, "reward_meter_mean": 0.45693451166152954, "reward_meter_std": 0.2026921510696411, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9970571994781494, "reward_repeat_soft_std": 0.002340377075597644, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.10925068706274033, "reward_total_composite_mean": 0.35164132714271545, "reward_total_composite_std": 0.22201529145240784} {"timestamp_utc": "2026-04-13T07:40:56Z", "mode": "train", "global_step": 142, "epoch": 0.014264188849824208, "loss": -0.0096, "grad_norm": 17.13520622253418, "learning_rate": 9.572727272727273e-06, "num_tokens": 264821.0, "completions/mean_length": 42.125, "completions/min_length": 37.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.125, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.6349161863327026, "rewards/meter/std": 0.3797772526741028, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9980885982513428, "rewards/repeat_soft/std": 0.0032931563910096884, "rewards/judge_quality/mean": 0.5987499952316284, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.6001944541931152, "rewards/total_composite/std": 0.17118269205093384, "reward": 0.6001944541931152, "reward_std": 0.17118267714977264, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2538120746612549, "sampling/sampling_logp_difference/max": 2.163132667541504, "sampling/importance_sampling_ratio/min": 0.11496441066265106, "sampling/importance_sampling_ratio/mean": 1.057732343673706, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.7320705354213715, "clip_ratio/low_mean": 0.0996027011424303, "clip_ratio/low_min": 0.0996027011424303, "clip_ratio/high_mean": 0.0937653873115778, "clip_ratio/high_max": 0.0937653873115778, "clip_ratio/region_mean": 0.1933680884540081, "reward_total_mean": 0.6001944541931152, "reward_meter_mean": 0.6349161863327026, "reward_meter_std": 0.3797772526741028, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9980885982513428, "reward_repeat_soft_std": 0.0032931563910096884, "reward_judge_quality_mean": 0.5987499952316284, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.6001944541931152, "reward_total_composite_std": 0.17118269205093384} {"timestamp_utc": "2026-04-13T07:41:04Z", "mode": "train", "global_step": 143, "epoch": 0.014364640883977901, "loss": 0.022, "grad_norm": 7.532354354858398, "learning_rate": 9.56969696969697e-06, "num_tokens": 267745.0, "completions/mean_length": 178.5, "completions/min_length": 127.0, "completions/max_length": 198.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 178.5, "completions/min_terminated_length": 127.0, "completions/max_terminated_length": 198.0, "rewards/meter/mean": 0.6103168725967407, "rewards/meter/std": 0.23132774233818054, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9940958023071289, "rewards/repeat_soft/std": 0.006265062373131514, "rewards/judge_quality/mean": 0.41499999165534973, "rewards/judge_quality/std": 0.14520922303199768, "rewards/total_composite/mean": 0.5102753043174744, "rewards/total_composite/std": 0.07604330033063889, "reward": 0.5102753043174744, "reward_std": 0.07604330778121948, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19552329182624817, "sampling/sampling_logp_difference/max": 1.7654690742492676, "sampling/importance_sampling_ratio/min": 0.1711065024137497, "sampling/importance_sampling_ratio/mean": 1.0306285619735718, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4039943516254425, "clip_ratio/low_mean": 0.12090220488607883, "clip_ratio/low_min": 0.12090220488607883, "clip_ratio/high_mean": 0.0771543812006712, "clip_ratio/high_max": 0.0771543812006712, "clip_ratio/region_mean": 0.19805658608675003, "reward_total_mean": 0.5102753043174744, "reward_meter_mean": 0.6103168725967407, "reward_meter_std": 0.23132774233818054, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9940958023071289, "reward_repeat_soft_std": 0.006265062373131514, "reward_judge_quality_mean": 0.41499999165534973, "reward_judge_quality_std": 0.14520922303199768, "reward_total_composite_mean": 0.5102753043174744, "reward_total_composite_std": 0.07604330033063889} {"timestamp_utc": "2026-04-13T07:41:11Z", "mode": "train", "global_step": 144, "epoch": 0.014465092918131592, "loss": 0.0646, "grad_norm": 16.326160430908203, "learning_rate": 9.566666666666668e-06, "num_tokens": 269457.0, "completions/mean_length": 42.0, "completions/min_length": 36.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.0, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.8048985004425049, "rewards/meter/std": 0.31660687923431396, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9979851245880127, "rewards/repeat_soft/std": 0.0019468758255243301, "rewards/judge_quality/mean": 0.5562499761581421, "rewards/judge_quality/std": 0.2249404937028885, "rewards/total_composite/mean": 0.6515523195266724, "rewards/total_composite/std": 0.18092253804206848, "reward": 0.6515523195266724, "reward_std": 0.1809225231409073, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2573746144771576, "sampling/sampling_logp_difference/max": 1.4676711559295654, "sampling/importance_sampling_ratio/min": 0.23046156764030457, "sampling/importance_sampling_ratio/mean": 1.0187652111053467, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.5685728192329407, "clip_ratio/low_mean": 0.14593254402279854, "clip_ratio/low_min": 0.14593254402279854, "clip_ratio/high_mean": 0.06349206529557705, "clip_ratio/high_max": 0.06349206529557705, "clip_ratio/region_mean": 0.2094246093183756, "reward_total_mean": 0.6515523195266724, "reward_meter_mean": 0.8048985004425049, "reward_meter_std": 0.31660687923431396, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9979851245880127, "reward_repeat_soft_std": 0.0019468758255243301, "reward_judge_quality_mean": 0.5562499761581421, "reward_judge_quality_std": 0.2249404937028885, "reward_total_composite_mean": 0.6515523195266724, "reward_total_composite_std": 0.18092253804206848} {"timestamp_utc": "2026-04-13T07:41:18Z", "mode": "train", "global_step": 145, "epoch": 0.014565544952285283, "loss": -0.0136, "grad_norm": 13.721324920654297, "learning_rate": 9.563636363636365e-06, "num_tokens": 271000.0, "completions/mean_length": 37.875, "completions/min_length": 32.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.875, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.41020357608795166, "rewards/meter/std": 0.468443363904953, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9902426600456238, "rewards/repeat_soft/std": 0.014534646645188332, "rewards/judge_quality/mean": 0.5824999809265137, "rewards/judge_quality/std": 0.23260943591594696, "rewards/total_composite/mean": 0.4963862895965576, "rewards/total_composite/std": 0.16369476914405823, "reward": 0.4963862895965576, "reward_std": 0.16369476914405823, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.25721344351768494, "sampling/sampling_logp_difference/max": 1.5890684127807617, "sampling/importance_sampling_ratio/min": 0.20411567389965057, "sampling/importance_sampling_ratio/mean": 1.0349671840667725, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1548607498407364, "clip_ratio/low_mean": 0.1378176361322403, "clip_ratio/low_min": 0.1378176361322403, "clip_ratio/high_mean": 0.08839784376323223, "clip_ratio/high_max": 0.08839784376323223, "clip_ratio/region_mean": 0.22621547989547253, "reward_total_mean": 0.4963862895965576, "reward_meter_mean": 0.41020357608795166, "reward_meter_std": 0.468443363904953, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9902426600456238, "reward_repeat_soft_std": 0.014534646645188332, "reward_judge_quality_mean": 0.5824999809265137, "reward_judge_quality_std": 0.23260943591594696, "reward_total_composite_mean": 0.4963862895965576, "reward_total_composite_std": 0.16369476914405823} {"timestamp_utc": "2026-04-13T07:41:25Z", "mode": "train", "global_step": 146, "epoch": 0.014665996986438976, "loss": 0.3095, "grad_norm": 20.189346313476562, "learning_rate": 9.56060606060606e-06, "num_tokens": 272617.0, "completions/mean_length": 33.125, "completions/min_length": 23.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.125, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.31975626945495605, "rewards/meter/std": 0.41625145077705383, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9912452697753906, "rewards/repeat_soft/std": 0.007847433909773827, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.3928127586841583, "rewards/total_composite/std": 0.19192352890968323, "reward": 0.3928127586841583, "reward_std": 0.19192351400852203, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24665847420692444, "sampling/sampling_logp_difference/max": 1.5526294708251953, "sampling/importance_sampling_ratio/min": 0.21169060468673706, "sampling/importance_sampling_ratio/mean": 1.0148438215255737, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.123029351234436, "clip_ratio/low_mean": 0.10523463413119316, "clip_ratio/low_min": 0.10523463413119316, "clip_ratio/high_mean": 0.13094667345285416, "clip_ratio/high_max": 0.13094667345285416, "clip_ratio/region_mean": 0.23618130758404732, "reward_total_mean": 0.3928127586841583, "reward_meter_mean": 0.31975626945495605, "reward_meter_std": 0.41625145077705383, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9912452697753906, "reward_repeat_soft_std": 0.007847433909773827, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.3928127586841583, "reward_total_composite_std": 0.19192352890968323} {"timestamp_utc": "2026-04-13T07:41:31Z", "mode": "train", "global_step": 147, "epoch": 0.014766449020592667, "loss": -0.007, "grad_norm": 22.95297622680664, "learning_rate": 9.55757575757576e-06, "num_tokens": 274100.0, "completions/mean_length": 28.375, "completions/min_length": 25.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.375, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.7506793141365051, "rewards/meter/std": 0.35030412673950195, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9963032007217407, "rewards/repeat_soft/std": 0.007101293187588453, "rewards/judge_quality/mean": 0.48125001788139343, "rewards/judge_quality/std": 0.22949868440628052, "rewards/total_composite/mean": 0.5625472068786621, "rewards/total_composite/std": 0.27261069416999817, "reward": 0.5625472068786621, "reward_std": 0.27261069416999817, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2450028359889984, "sampling/sampling_logp_difference/max": 1.6066884994506836, "sampling/importance_sampling_ratio/min": 0.2005506455898285, "sampling/importance_sampling_ratio/mean": 0.9905889630317688, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7993857562541962, "clip_ratio/low_mean": 0.05597480107098818, "clip_ratio/low_min": 0.05597480107098818, "clip_ratio/high_mean": 0.15836711786687374, "clip_ratio/high_max": 0.15836711786687374, "clip_ratio/region_mean": 0.21434191893786192, "reward_total_mean": 0.5625472068786621, "reward_meter_mean": 0.7506793141365051, "reward_meter_std": 0.35030412673950195, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9963032007217407, "reward_repeat_soft_std": 0.007101293187588453, "reward_judge_quality_mean": 0.48125001788139343, "reward_judge_quality_std": 0.22949868440628052, "reward_total_composite_mean": 0.5625472068786621, "reward_total_composite_std": 0.27261069416999817} {"timestamp_utc": "2026-04-13T07:41:40Z", "mode": "train", "global_step": 148, "epoch": 0.014866901054746359, "loss": -0.0024, "grad_norm": 11.477031707763672, "learning_rate": 9.554545454545455e-06, "num_tokens": 276455.0, "completions/mean_length": 105.375, "completions/min_length": 86.0, "completions/max_length": 164.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 105.375, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 164.0, "rewards/meter/mean": 0.7320038080215454, "rewards/meter/std": 0.3469865918159485, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9950052499771118, "rewards/repeat_soft/std": 0.003556472947821021, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.5532875061035156, "rewards/total_composite/std": 0.08732420206069946, "reward": 0.5532875061035156, "reward_std": 0.08732419461011887, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21940535306930542, "sampling/sampling_logp_difference/max": 2.2431559562683105, "sampling/importance_sampling_ratio/min": 0.10612305998802185, "sampling/importance_sampling_ratio/mean": 1.0478465557098389, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.533340662717819, "clip_ratio/low_mean": 0.07008165493607521, "clip_ratio/low_min": 0.07008165493607521, "clip_ratio/high_mean": 0.15172532014548779, "clip_ratio/high_max": 0.15172532014548779, "clip_ratio/region_mean": 0.221806975081563, "reward_total_mean": 0.5532875061035156, "reward_meter_mean": 0.7320038080215454, "reward_meter_std": 0.3469865918159485, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9950052499771118, "reward_repeat_soft_std": 0.003556472947821021, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.5532875061035156, "reward_total_composite_std": 0.08732420206069946} {"timestamp_utc": "2026-04-13T07:41:49Z", "mode": "train", "global_step": 149, "epoch": 0.01496735308890005, "loss": 0.114, "grad_norm": 7.869264602661133, "learning_rate": 9.551515151515152e-06, "num_tokens": 279217.0, "completions/mean_length": 154.25, "completions/min_length": 118.0, "completions/max_length": 219.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 154.25, "completions/min_terminated_length": 118.0, "completions/max_terminated_length": 219.0, "rewards/meter/mean": 0.9051262140274048, "rewards/meter/std": 0.10090532898902893, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9958059787750244, "rewards/repeat_soft/std": 0.0055671576410532, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.5140024423599243, "rewards/total_composite/std": 0.21084751188755035, "reward": 0.5140024423599243, "reward_std": 0.21084751188755035, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22774766385555267, "sampling/sampling_logp_difference/max": 1.7449188232421875, "sampling/importance_sampling_ratio/min": 0.17465916275978088, "sampling/importance_sampling_ratio/mean": 1.0631458759307861, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.3837957084178925, "clip_ratio/low_mean": 0.03640520200133324, "clip_ratio/low_min": 0.03640520200133324, "clip_ratio/high_mean": 0.1736424770206213, "clip_ratio/high_max": 0.1736424770206213, "clip_ratio/region_mean": 0.21004767902195454, "reward_total_mean": 0.5140024423599243, "reward_meter_mean": 0.9051262140274048, "reward_meter_std": 0.10090532898902893, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9958059787750244, "reward_repeat_soft_std": 0.0055671576410532, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.5140024423599243, "reward_total_composite_std": 0.21084751188755035} {"timestamp_utc": "2026-04-13T07:41:57Z", "mode": "train", "global_step": 150, "epoch": 0.015067805123053743, "loss": 0.0883, "grad_norm": 12.227827072143555, "learning_rate": 9.54848484848485e-06, "num_tokens": 281338.0, "completions/mean_length": 95.125, "completions/min_length": 70.0, "completions/max_length": 112.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.125, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 112.0, "rewards/meter/mean": 0.3991343379020691, "rewards/meter/std": 0.34587615728378296, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.991938591003418, "rewards/repeat_soft/std": 0.006831905338913202, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4577544927597046, "rewards/total_composite/std": 0.09414715319871902, "reward": 0.4577544927597046, "reward_std": 0.09414716064929962, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2518637478351593, "sampling/sampling_logp_difference/max": 2.145904541015625, "sampling/importance_sampling_ratio/min": 0.11696219444274902, "sampling/importance_sampling_ratio/mean": 1.0258044004440308, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.635319620370865, "clip_ratio/low_mean": 0.13855895400047302, "clip_ratio/low_min": 0.13855895400047302, "clip_ratio/high_mean": 0.09044339880347252, "clip_ratio/high_max": 0.09044339880347252, "clip_ratio/region_mean": 0.22900235280394554, "reward_total_mean": 0.4577544927597046, "reward_meter_mean": 0.3991343379020691, "reward_meter_std": 0.34587615728378296, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.991938591003418, "reward_repeat_soft_std": 0.006831905338913202, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4577544927597046, "reward_total_composite_std": 0.09414715319871902} {"timestamp_utc": "2026-04-13T07:42:46Z", "mode": "eval", "global_step": 150, "epoch": 0.015067805123053743, "eval_loss": NaN, "eval_runtime": 49.5893, "eval_samples_per_second": 1.613, "eval_steps_per_second": 0.202, "eval_num_tokens": 281338.0, "eval_completions/mean_length": 72.475, "eval_completions/min_length": 28.6, "eval_completions/max_length": 138.9, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 72.475, "eval_completions/min_terminated_length": 28.6, "eval_completions/max_terminated_length": 138.9, "eval_rewards/meter/mean": 0.5251416712999344, "eval_rewards/meter/std": 0.39672465324401857, "eval_rewards/count_adherence/mean": 0.9891666531562805, "eval_rewards/count_adherence/std": 0.03064129650592804, "eval_rewards/hard_gate/mean": 0.9625, "eval_rewards/hard_gate/std": 0.0816463440656662, "eval_rewards/repeat_soft/mean": 0.9936495900154114, "eval_rewards/repeat_soft/std": 0.00908708639908582, "eval_rewards/judge_quality/mean": 0.4742499977350235, "eval_rewards/judge_quality/std": 0.17209785655140877, "eval_rewards/total_composite/mean": 0.5050697594881057, "eval_rewards/total_composite/std": 0.17620564475655556, "eval_reward": 0.5050697594881057, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.1579168274998665, "eval_sampling/sampling_logp_difference/max": 1.197763156890869, "eval_sampling/importance_sampling_ratio/min": 0.3089745670557022, "eval_sampling/importance_sampling_ratio/mean": 1.0503769397735596, "eval_sampling/importance_sampling_ratio/max": 1.5593770742416382, "eval_entropy": 2.493303906917572, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5050697594881057, "eval_reward_meter_mean": 0.5251416712999344, "eval_reward_meter_std": 0.39672465324401857, "eval_reward_count_adherence_mean": 0.9891666531562805, "eval_reward_count_adherence_std": 0.03064129650592804, "eval_reward_hard_gate_mean": 0.9625, "eval_reward_hard_gate_std": 0.0816463440656662, "eval_reward_repeat_soft_mean": 0.9936495900154114, "eval_reward_repeat_soft_std": 0.00908708639908582, "eval_reward_judge_quality_mean": 0.4742499977350235, "eval_reward_judge_quality_std": 0.17209785655140877, "eval_reward_total_composite_mean": 0.5050697594881057, "eval_reward_total_composite_std": 0.17620564475655556} {"timestamp_utc": "2026-04-13T07:42:55Z", "mode": "train", "global_step": 151, "epoch": 0.015168257157207434, "loss": -0.0139, "grad_norm": 19.790287017822266, "learning_rate": 9.545454545454547e-06, "num_tokens": 282777.0, "completions/mean_length": 27.875, "completions/min_length": 20.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.875, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.6628537178039551, "rewards/meter/std": 0.2736438810825348, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9575977325439453, "rewards/repeat_soft/std": 0.013865554705262184, "rewards/judge_quality/mean": 0.41749998927116394, "rewards/judge_quality/std": 0.06902380287647247, "rewards/total_composite/mean": 0.5267788767814636, "rewards/total_composite/std": 0.08878375589847565, "reward": 0.5267788767814636, "reward_std": 0.08878375589847565, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19810548424720764, "sampling/sampling_logp_difference/max": 1.0578088760375977, "sampling/importance_sampling_ratio/min": 0.34721577167510986, "sampling/importance_sampling_ratio/mean": 1.0431057214736938, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.909252405166626, "clip_ratio/low_mean": 0.08024779241532087, "clip_ratio/low_min": 0.08024779241532087, "clip_ratio/high_mean": 0.08133874274790287, "clip_ratio/high_max": 0.08133874274790287, "clip_ratio/region_mean": 0.16158653516322374, "reward_total_mean": 0.5267788767814636, "reward_meter_mean": 0.6628537178039551, "reward_meter_std": 0.2736438810825348, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9575977325439453, "reward_repeat_soft_std": 0.013865554705262184, "reward_judge_quality_mean": 0.41749998927116394, "reward_judge_quality_std": 0.06902380287647247, "reward_total_composite_mean": 0.5267788767814636, "reward_total_composite_std": 0.08878375589847565} {"timestamp_utc": "2026-04-13T07:43:04Z", "mode": "train", "global_step": 152, "epoch": 0.015268709191361125, "loss": -0.0552, "grad_norm": 9.100715637207031, "learning_rate": 9.542424242424242e-06, "num_tokens": 284947.0, "completions/mean_length": 108.25, "completions/min_length": 66.0, "completions/max_length": 145.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.25, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.696089506149292, "rewards/meter/std": 0.368598073720932, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.996252179145813, "rewards/repeat_soft/std": 0.0039284187369048595, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.39426562190055847, "rewards/total_composite/std": 0.25745531916618347, "reward": 0.39426562190055847, "reward_std": 0.2574552893638611, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23037481307983398, "sampling/sampling_logp_difference/max": 1.9209213256835938, "sampling/importance_sampling_ratio/min": 0.14647194743156433, "sampling/importance_sampling_ratio/mean": 1.0479086637496948, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.9769923985004425, "clip_ratio/low_mean": 0.07217737101018429, "clip_ratio/low_min": 0.07217737101018429, "clip_ratio/high_mean": 0.13639610446989536, "clip_ratio/high_max": 0.13639610446989536, "clip_ratio/region_mean": 0.20857347548007965, "reward_total_mean": 0.39426562190055847, "reward_meter_mean": 0.696089506149292, "reward_meter_std": 0.368598073720932, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.996252179145813, "reward_repeat_soft_std": 0.0039284187369048595, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.39426562190055847, "reward_total_composite_std": 0.25745531916618347} {"timestamp_utc": "2026-04-13T07:43:11Z", "mode": "train", "global_step": 153, "epoch": 0.015369161225514816, "loss": 0.0375, "grad_norm": 22.370010375976562, "learning_rate": 9.539393939393941e-06, "num_tokens": 286482.0, "completions/mean_length": 45.875, "completions/min_length": 40.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.875, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.7622062563896179, "rewards/meter/std": 0.33447155356407166, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9981516599655151, "rewards/repeat_soft/std": 0.0038459571078419685, "rewards/judge_quality/mean": 0.4424999952316284, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.5700531005859375, "rewards/total_composite/std": 0.09727941453456879, "reward": 0.5700531005859375, "reward_std": 0.09727941453456879, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21602395176887512, "sampling/sampling_logp_difference/max": 1.35379958152771, "sampling/importance_sampling_ratio/min": 0.258257120847702, "sampling/importance_sampling_ratio/mean": 1.035374402999878, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.340231239795685, "clip_ratio/low_mean": 0.06541666761040688, "clip_ratio/low_min": 0.06541666761040688, "clip_ratio/high_mean": 0.10533494502305984, "clip_ratio/high_max": 0.10533494502305984, "clip_ratio/region_mean": 0.17075161263346672, "reward_total_mean": 0.5700531005859375, "reward_meter_mean": 0.7622062563896179, "reward_meter_std": 0.33447155356407166, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9981516599655151, "reward_repeat_soft_std": 0.0038459571078419685, "reward_judge_quality_mean": 0.4424999952316284, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.5700531005859375, "reward_total_composite_std": 0.09727941453456879} {"timestamp_utc": "2026-04-13T07:43:18Z", "mode": "train", "global_step": 154, "epoch": 0.015469613259668509, "loss": -0.0714, "grad_norm": 16.674734115600586, "learning_rate": 9.536363636363637e-06, "num_tokens": 288210.0, "completions/mean_length": 44.0, "completions/min_length": 26.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.0, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.37179723381996155, "rewards/meter/std": 0.3616820275783539, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9980150461196899, "rewards/repeat_soft/std": 0.004647395573556423, "rewards/judge_quality/mean": 0.35499998927116394, "rewards/judge_quality/std": 0.11928357183933258, "rewards/total_composite/mean": 0.4344622492790222, "rewards/total_composite/std": 0.09572197496891022, "reward": 0.4344622492790222, "reward_std": 0.09572196751832962, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.247511625289917, "sampling/sampling_logp_difference/max": 1.6972360610961914, "sampling/importance_sampling_ratio/min": 0.18318915367126465, "sampling/importance_sampling_ratio/mean": 1.0838520526885986, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.3497215807437897, "clip_ratio/low_mean": 0.12646591290831566, "clip_ratio/low_min": 0.12646591290831566, "clip_ratio/high_mean": 0.0763119999319315, "clip_ratio/high_max": 0.0763119999319315, "clip_ratio/region_mean": 0.20277791284024715, "reward_total_mean": 0.4344622492790222, "reward_meter_mean": 0.37179723381996155, "reward_meter_std": 0.3616820275783539, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9980150461196899, "reward_repeat_soft_std": 0.004647395573556423, "reward_judge_quality_mean": 0.35499998927116394, "reward_judge_quality_std": 0.11928357183933258, "reward_total_composite_mean": 0.4344622492790222, "reward_total_composite_std": 0.09572197496891022} {"timestamp_utc": "2026-04-13T07:43:24Z", "mode": "train", "global_step": 155, "epoch": 0.0155700652938222, "loss": 0.1263, "grad_norm": 13.679117202758789, "learning_rate": 9.533333333333334e-06, "num_tokens": 290021.0, "completions/mean_length": 50.375, "completions/min_length": 33.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.375, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.3696790039539337, "rewards/meter/std": 0.3822711706161499, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9926376342773438, "rewards/repeat_soft/std": 0.012575971893966198, "rewards/judge_quality/mean": 0.3999999761581421, "rewards/judge_quality/std": 0.09258200973272324, "rewards/total_composite/mean": 0.4402022957801819, "rewards/total_composite/std": 0.12922914326190948, "reward": 0.4402022957801819, "reward_std": 0.12922914326190948, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22571733593940735, "sampling/sampling_logp_difference/max": 1.248558521270752, "sampling/importance_sampling_ratio/min": 0.28691810369491577, "sampling/importance_sampling_ratio/mean": 1.0440168380737305, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.9790842533111572, "clip_ratio/low_mean": 0.09155844338238239, "clip_ratio/low_min": 0.09155844338238239, "clip_ratio/high_mean": 0.09837852418422699, "clip_ratio/high_max": 0.09837852418422699, "clip_ratio/region_mean": 0.18993696756660938, "reward_total_mean": 0.4402022957801819, "reward_meter_mean": 0.3696790039539337, "reward_meter_std": 0.3822711706161499, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9926376342773438, "reward_repeat_soft_std": 0.012575971893966198, "reward_judge_quality_mean": 0.3999999761581421, "reward_judge_quality_std": 0.09258200973272324, "reward_total_composite_mean": 0.4402022957801819, "reward_total_composite_std": 0.12922914326190948} {"timestamp_utc": "2026-04-13T07:43:31Z", "mode": "train", "global_step": 156, "epoch": 0.01567051732797589, "loss": -0.0267, "grad_norm": 19.237661361694336, "learning_rate": 9.530303030303031e-06, "num_tokens": 291601.0, "completions/mean_length": 34.5, "completions/min_length": 30.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.5, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.7581612467765808, "rewards/meter/std": 0.3878549337387085, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9873794317245483, "rewards/repeat_soft/std": 0.028771191835403442, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.562809944152832, "rewards/total_composite/std": 0.10876806825399399, "reward": 0.562809944152832, "reward_std": 0.10876806825399399, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21949085593223572, "sampling/sampling_logp_difference/max": 1.4842844009399414, "sampling/importance_sampling_ratio/min": 0.2266644984483719, "sampling/importance_sampling_ratio/mean": 1.0325417518615723, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2725380212068558, "clip_ratio/low_mean": 0.05151515267789364, "clip_ratio/low_min": 0.05151515267789364, "clip_ratio/high_mean": 0.14134271629154682, "clip_ratio/high_max": 0.14134271629154682, "clip_ratio/region_mean": 0.19285786896944046, "reward_total_mean": 0.562809944152832, "reward_meter_mean": 0.7581612467765808, "reward_meter_std": 0.3878549337387085, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9873794317245483, "reward_repeat_soft_std": 0.028771191835403442, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.562809944152832, "reward_total_composite_std": 0.10876806825399399} {"timestamp_utc": "2026-04-13T07:43:38Z", "mode": "train", "global_step": 157, "epoch": 0.015770969362129583, "loss": 0.0124, "grad_norm": 18.231548309326172, "learning_rate": 9.527272727272729e-06, "num_tokens": 293024.0, "completions/mean_length": 24.875, "completions/min_length": 17.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.875, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.484775185585022, "rewards/meter/std": 0.4647696018218994, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.961837112903595, "rewards/repeat_soft/std": 0.0018749026348814368, "rewards/judge_quality/mean": 0.3512499928474426, "rewards/judge_quality/std": 0.11630471050739288, "rewards/total_composite/mean": 0.4474857449531555, "rewards/total_composite/std": 0.11991287022829056, "reward": 0.4474857449531555, "reward_std": 0.11991287022829056, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2207426279783249, "sampling/sampling_logp_difference/max": 1.7150442600250244, "sampling/importance_sampling_ratio/min": 0.17995575070381165, "sampling/importance_sampling_ratio/mean": 1.051077961921692, "sampling/importance_sampling_ratio/max": 1.9085829257965088, "entropy": 2.074438989162445, "clip_ratio/low_mean": 0.117631989531219, "clip_ratio/low_min": 0.117631989531219, "clip_ratio/high_mean": 0.09234270825982094, "clip_ratio/high_max": 0.09234270825982094, "clip_ratio/region_mean": 0.20997469779103994, "reward_total_mean": 0.4474857449531555, "reward_meter_mean": 0.484775185585022, "reward_meter_std": 0.4647696018218994, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.961837112903595, "reward_repeat_soft_std": 0.0018749026348814368, "reward_judge_quality_mean": 0.3512499928474426, "reward_judge_quality_std": 0.11630471050739288, "reward_total_composite_mean": 0.4474857449531555, "reward_total_composite_std": 0.11991287022829056} {"timestamp_utc": "2026-04-13T07:43:46Z", "mode": "train", "global_step": 158, "epoch": 0.015871421396283274, "loss": 0.3052, "grad_norm": 15.065874099731445, "learning_rate": 9.524242424242424e-06, "num_tokens": 294583.0, "completions/mean_length": 46.875, "completions/min_length": 33.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.875, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.8994266986846924, "rewards/meter/std": 0.15715087950229645, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9950315952301025, "rewards/repeat_soft/std": 0.006452750880271196, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.5939333438873291, "rewards/total_composite/std": 0.11897853761911392, "reward": 0.5939333438873291, "reward_std": 0.11897853016853333, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2199499011039734, "sampling/sampling_logp_difference/max": 1.2576894760131836, "sampling/importance_sampling_ratio/min": 0.28431016206741333, "sampling/importance_sampling_ratio/mean": 1.0469788312911987, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.943353235721588, "clip_ratio/low_mean": 0.07501471135765314, "clip_ratio/low_min": 0.07501471135765314, "clip_ratio/high_mean": 0.14738421700894833, "clip_ratio/high_max": 0.14738421700894833, "clip_ratio/region_mean": 0.22239892836660147, "reward_total_mean": 0.5939333438873291, "reward_meter_mean": 0.8994266986846924, "reward_meter_std": 0.15715087950229645, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9950315952301025, "reward_repeat_soft_std": 0.006452750880271196, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.5939333438873291, "reward_total_composite_std": 0.11897853761911392} {"timestamp_utc": "2026-04-13T07:43:53Z", "mode": "train", "global_step": 159, "epoch": 0.015971873430436965, "loss": 0.0037, "grad_norm": 13.692737579345703, "learning_rate": 9.521212121212121e-06, "num_tokens": 296129.0, "completions/mean_length": 40.25, "completions/min_length": 36.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.25, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.8731591701507568, "rewards/meter/std": 0.2492521107196808, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9962665438652039, "rewards/repeat_soft/std": 0.004987684544175863, "rewards/judge_quality/mean": 0.4724999964237213, "rewards/judge_quality/std": 0.1011011004447937, "rewards/total_composite/mean": 0.6194087266921997, "rewards/total_composite/std": 0.10308452695608139, "reward": 0.6194087266921997, "reward_std": 0.10308452695608139, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21989428997039795, "sampling/sampling_logp_difference/max": 2.073795795440674, "sampling/importance_sampling_ratio/min": 0.1257077157497406, "sampling/importance_sampling_ratio/mean": 1.0411407947540283, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.5392094254493713, "clip_ratio/low_mean": 0.07449252251535654, "clip_ratio/low_min": 0.07449252251535654, "clip_ratio/high_mean": 0.13062131218612194, "clip_ratio/high_max": 0.13062131218612194, "clip_ratio/region_mean": 0.20511383470147848, "reward_total_mean": 0.6194087266921997, "reward_meter_mean": 0.8731591701507568, "reward_meter_std": 0.2492521107196808, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9962665438652039, "reward_repeat_soft_std": 0.004987684544175863, "reward_judge_quality_mean": 0.4724999964237213, "reward_judge_quality_std": 0.1011011004447937, "reward_total_composite_mean": 0.6194087266921997, "reward_total_composite_std": 0.10308452695608139} {"timestamp_utc": "2026-04-13T07:43:59Z", "mode": "train", "global_step": 160, "epoch": 0.01607232546459066, "loss": -0.0457, "grad_norm": 20.655656814575195, "learning_rate": 9.518181818181819e-06, "num_tokens": 297670.0, "completions/mean_length": 35.625, "completions/min_length": 27.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.625, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.730107843875885, "rewards/meter/std": 0.3464305102825165, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9918180704116821, "rewards/repeat_soft/std": 0.02118588052690029, "rewards/judge_quality/mean": 0.70250004529953, "rewards/judge_quality/std": 0.17862571775913239, "rewards/total_composite/mean": 0.6042086482048035, "rewards/total_composite/std": 0.3024638891220093, "reward": 0.6042086482048035, "reward_std": 0.3024638593196869, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20736154913902283, "sampling/sampling_logp_difference/max": 3.032355308532715, "sampling/importance_sampling_ratio/min": 0.04820197448134422, "sampling/importance_sampling_ratio/mean": 1.0262166261672974, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.698737993836403, "clip_ratio/low_mean": 0.07620247639715672, "clip_ratio/low_min": 0.07620247639715672, "clip_ratio/high_mean": 0.10640341974794865, "clip_ratio/high_max": 0.10640341974794865, "clip_ratio/region_mean": 0.18260589614510536, "reward_total_mean": 0.6042086482048035, "reward_meter_mean": 0.730107843875885, "reward_meter_std": 0.3464305102825165, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9918180704116821, "reward_repeat_soft_std": 0.02118588052690029, "reward_judge_quality_mean": 0.70250004529953, "reward_judge_quality_std": 0.17862571775913239, "reward_total_composite_mean": 0.6042086482048035, "reward_total_composite_std": 0.3024638891220093} {"timestamp_utc": "2026-04-13T07:44:06Z", "mode": "train", "global_step": 161, "epoch": 0.01617277749874435, "loss": 0.0961, "grad_norm": 20.642906188964844, "learning_rate": 9.515151515151516e-06, "num_tokens": 299034.0, "completions/mean_length": 24.5, "completions/min_length": 19.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.5, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.8733356595039368, "rewards/meter/std": 0.15982545912265778, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9569480419158936, "rewards/repeat_soft/std": 0.010414325632154942, "rewards/judge_quality/mean": 0.5724999904632568, "rewards/judge_quality/std": 0.38425251841545105, "rewards/total_composite/mean": 0.6833223104476929, "rewards/total_composite/std": 0.2551179528236389, "reward": 0.6833223104476929, "reward_std": 0.2551179528236389, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21364142000675201, "sampling/sampling_logp_difference/max": 1.458907127380371, "sampling/importance_sampling_ratio/min": 0.23249022662639618, "sampling/importance_sampling_ratio/mean": 1.0699632167816162, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2969636619091034, "clip_ratio/low_mean": 0.08534770319238305, "clip_ratio/low_min": 0.08534770319238305, "clip_ratio/high_mean": 0.07862903131172061, "clip_ratio/high_max": 0.07862903131172061, "clip_ratio/region_mean": 0.16397673450410366, "reward_total_mean": 0.6833223104476929, "reward_meter_mean": 0.8733356595039368, "reward_meter_std": 0.15982545912265778, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9569480419158936, "reward_repeat_soft_std": 0.010414325632154942, "reward_judge_quality_mean": 0.5724999904632568, "reward_judge_quality_std": 0.38425251841545105, "reward_total_composite_mean": 0.6833223104476929, "reward_total_composite_std": 0.2551179528236389} {"timestamp_utc": "2026-04-13T07:44:13Z", "mode": "train", "global_step": 162, "epoch": 0.016273229532898042, "loss": -0.0122, "grad_norm": 25.833316802978516, "learning_rate": 9.512121212121213e-06, "num_tokens": 300545.0, "completions/mean_length": 28.875, "completions/min_length": 22.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.875, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.5683460235595703, "rewards/meter/std": 0.4032154679298401, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9980363845825195, "rewards/repeat_soft/std": 0.004261462949216366, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.509473979473114, "rewards/total_composite/std": 0.11501184850931168, "reward": 0.509473979473114, "reward_std": 0.11501184850931168, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2363111972808838, "sampling/sampling_logp_difference/max": 2.019106864929199, "sampling/importance_sampling_ratio/min": 0.1327739953994751, "sampling/importance_sampling_ratio/mean": 1.0371768474578857, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1397273391485214, "clip_ratio/low_mean": 0.09411422163248062, "clip_ratio/low_min": 0.09411422163248062, "clip_ratio/high_mean": 0.13323844783008099, "clip_ratio/high_max": 0.13323844783008099, "clip_ratio/region_mean": 0.2273526694625616, "reward_total_mean": 0.509473979473114, "reward_meter_mean": 0.5683460235595703, "reward_meter_std": 0.4032154679298401, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9980363845825195, "reward_repeat_soft_std": 0.004261462949216366, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.509473979473114, "reward_total_composite_std": 0.11501184850931168} {"timestamp_utc": "2026-04-13T07:44:20Z", "mode": "train", "global_step": 163, "epoch": 0.016373681567051733, "loss": 0.167, "grad_norm": 16.014175415039062, "learning_rate": 9.50909090909091e-06, "num_tokens": 302176.0, "completions/mean_length": 43.875, "completions/min_length": 28.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.875, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.7519853115081787, "rewards/meter/std": 0.37167438864707947, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9982995986938477, "rewards/repeat_soft/std": 0.0027445803862065077, "rewards/judge_quality/mean": 0.5712500214576721, "rewards/judge_quality/std": 0.24468272924423218, "rewards/total_composite/mean": 0.5745153427124023, "rewards/total_composite/std": 0.3193117678165436, "reward": 0.5745153427124023, "reward_std": 0.3193117678165436, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2562037706375122, "sampling/sampling_logp_difference/max": 2.644084930419922, "sampling/importance_sampling_ratio/min": 0.0710703581571579, "sampling/importance_sampling_ratio/mean": 1.032947301864624, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6925185918807983, "clip_ratio/low_mean": 0.09780711121857166, "clip_ratio/low_min": 0.09780711121857166, "clip_ratio/high_mean": 0.10800338350236416, "clip_ratio/high_max": 0.10800338350236416, "clip_ratio/region_mean": 0.20581049472093582, "reward_total_mean": 0.5745153427124023, "reward_meter_mean": 0.7519853115081787, "reward_meter_std": 0.37167438864707947, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9982995986938477, "reward_repeat_soft_std": 0.0027445803862065077, "reward_judge_quality_mean": 0.5712500214576721, "reward_judge_quality_std": 0.24468272924423218, "reward_total_composite_mean": 0.5745153427124023, "reward_total_composite_std": 0.3193117678165436} {"timestamp_utc": "2026-04-13T07:44:27Z", "mode": "train", "global_step": 164, "epoch": 0.016474133601205424, "loss": 0.006, "grad_norm": 10.29527473449707, "learning_rate": 9.506060606060606e-06, "num_tokens": 304486.0, "completions/mean_length": 101.75, "completions/min_length": 73.0, "completions/max_length": 141.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 101.75, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 141.0, "rewards/meter/mean": 0.5555546879768372, "rewards/meter/std": 0.39578020572662354, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9981604814529419, "rewards/repeat_soft/std": 0.001627032645046711, "rewards/judge_quality/mean": 0.3812499940395355, "rewards/judge_quality/std": 0.17266297340393066, "rewards/total_composite/mean": 0.5006445646286011, "rewards/total_composite/std": 0.1136239543557167, "reward": 0.5006445646286011, "reward_std": 0.1136239543557167, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23732200264930725, "sampling/sampling_logp_difference/max": 1.4951133728027344, "sampling/importance_sampling_ratio/min": 0.22422318160533905, "sampling/importance_sampling_ratio/mean": 1.0576616525650024, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.6199953258037567, "clip_ratio/low_mean": 0.08103533275425434, "clip_ratio/low_min": 0.08103533275425434, "clip_ratio/high_mean": 0.14940478280186653, "clip_ratio/high_max": 0.14940478280186653, "clip_ratio/region_mean": 0.23044011555612087, "reward_total_mean": 0.5006445646286011, "reward_meter_mean": 0.5555546879768372, "reward_meter_std": 0.39578020572662354, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9981604814529419, "reward_repeat_soft_std": 0.001627032645046711, "reward_judge_quality_mean": 0.3812499940395355, "reward_judge_quality_std": 0.17266297340393066, "reward_total_composite_mean": 0.5006445646286011, "reward_total_composite_std": 0.1136239543557167} {"timestamp_utc": "2026-04-13T07:44:34Z", "mode": "train", "global_step": 165, "epoch": 0.016574585635359115, "loss": 0.2365, "grad_norm": 22.70522117614746, "learning_rate": 9.503030303030303e-06, "num_tokens": 306042.0, "completions/mean_length": 36.5, "completions/min_length": 23.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.5, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.49694645404815674, "rewards/meter/std": 0.3599587380886078, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9899628162384033, "rewards/repeat_soft/std": 0.015568912029266357, "rewards/judge_quality/mean": 0.6737500429153442, "rewards/judge_quality/std": 0.263435423374176, "rewards/total_composite/mean": 0.5811421275138855, "rewards/total_composite/std": 0.21729491651058197, "reward": 0.5811421275138855, "reward_std": 0.21729491651058197, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2418246865272522, "sampling/sampling_logp_difference/max": 3.4209160804748535, "sampling/importance_sampling_ratio/min": 0.03268248215317726, "sampling/importance_sampling_ratio/mean": 0.9846088886260986, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0012380257248878, "clip_ratio/low_mean": 0.11417735926806927, "clip_ratio/low_min": 0.11417735926806927, "clip_ratio/high_mean": 0.09063545241951942, "clip_ratio/high_max": 0.09063545241951942, "clip_ratio/region_mean": 0.2048128116875887, "reward_total_mean": 0.5811421275138855, "reward_meter_mean": 0.49694645404815674, "reward_meter_std": 0.3599587380886078, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9899628162384033, "reward_repeat_soft_std": 0.015568912029266357, "reward_judge_quality_mean": 0.6737500429153442, "reward_judge_quality_std": 0.263435423374176, "reward_total_composite_mean": 0.5811421275138855, "reward_total_composite_std": 0.21729491651058197} {"timestamp_utc": "2026-04-13T07:44:42Z", "mode": "train", "global_step": 166, "epoch": 0.016675037669512806, "loss": 0.1229, "grad_norm": 14.87144947052002, "learning_rate": 9.5e-06, "num_tokens": 307629.0, "completions/mean_length": 34.375, "completions/min_length": 24.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.375, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.5064627528190613, "rewards/meter/std": 0.37647876143455505, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9944350123405457, "rewards/repeat_soft/std": 0.004966493230313063, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.10507651418447495, "rewards/total_composite/mean": 0.5132936239242554, "rewards/total_composite/std": 0.15036630630493164, "reward": 0.5132936239242554, "reward_std": 0.15036627650260925, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22471682727336884, "sampling/sampling_logp_difference/max": 1.9718409776687622, "sampling/importance_sampling_ratio/min": 0.1392003446817398, "sampling/importance_sampling_ratio/mean": 1.0255013704299927, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8131614699959755, "clip_ratio/low_mean": 0.11302317772060633, "clip_ratio/low_min": 0.11302317772060633, "clip_ratio/high_mean": 0.08667140640318394, "clip_ratio/high_max": 0.08667140640318394, "clip_ratio/region_mean": 0.19969458412379026, "reward_total_mean": 0.5132936239242554, "reward_meter_mean": 0.5064627528190613, "reward_meter_std": 0.37647876143455505, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9944350123405457, "reward_repeat_soft_std": 0.004966493230313063, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.10507651418447495, "reward_total_composite_mean": 0.5132936239242554, "reward_total_composite_std": 0.15036630630493164} {"timestamp_utc": "2026-04-13T07:44:49Z", "mode": "train", "global_step": 167, "epoch": 0.016775489703666498, "loss": 0.0906, "grad_norm": 14.257298469543457, "learning_rate": 9.496969696969698e-06, "num_tokens": 309198.0, "completions/mean_length": 49.125, "completions/min_length": 34.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.125, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.6849068999290466, "rewards/meter/std": 0.4111604392528534, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.998062789440155, "rewards/repeat_soft/std": 0.00547928037121892, "rewards/judge_quality/mean": 0.42249998450279236, "rewards/judge_quality/std": 0.15471172332763672, "rewards/total_composite/mean": 0.46477779746055603, "rewards/total_composite/std": 0.30610400438308716, "reward": 0.46477779746055603, "reward_std": 0.30610397458076477, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22429433465003967, "sampling/sampling_logp_difference/max": 1.3679618835449219, "sampling/importance_sampling_ratio/min": 0.2546253800392151, "sampling/importance_sampling_ratio/mean": 1.047497272491455, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.0990798473358154, "clip_ratio/low_mean": 0.07017498183995485, "clip_ratio/low_min": 0.07017498183995485, "clip_ratio/high_mean": 0.12275388929992914, "clip_ratio/high_max": 0.12275388929992914, "clip_ratio/region_mean": 0.192928871139884, "reward_total_mean": 0.46477779746055603, "reward_meter_mean": 0.6849068999290466, "reward_meter_std": 0.4111604392528534, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.998062789440155, "reward_repeat_soft_std": 0.00547928037121892, "reward_judge_quality_mean": 0.42249998450279236, "reward_judge_quality_std": 0.15471172332763672, "reward_total_composite_mean": 0.46477779746055603, "reward_total_composite_std": 0.30610400438308716} {"timestamp_utc": "2026-04-13T07:44:56Z", "mode": "train", "global_step": 168, "epoch": 0.016875941737820192, "loss": 0.016, "grad_norm": 24.050216674804688, "learning_rate": 9.493939393939395e-06, "num_tokens": 310708.0, "completions/mean_length": 23.75, "completions/min_length": 18.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.75, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.7325787544250488, "rewards/meter/std": 0.39579838514328003, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.961837112903595, "rewards/repeat_soft/std": 0.0018749026348814368, "rewards/judge_quality/mean": 0.36374998092651367, "rewards/judge_quality/std": 0.13265825808048248, "rewards/total_composite/mean": 0.4768216609954834, "rewards/total_composite/std": 0.21741239726543427, "reward": 0.4768216609954834, "reward_std": 0.21741239726543427, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22347064316272736, "sampling/sampling_logp_difference/max": 1.902097225189209, "sampling/importance_sampling_ratio/min": 0.14925527572631836, "sampling/importance_sampling_ratio/mean": 1.030205249786377, "sampling/importance_sampling_ratio/max": 1.782066822052002, "entropy": 1.9697849974036217, "clip_ratio/low_mean": 0.09212963096797466, "clip_ratio/low_min": 0.09212963096797466, "clip_ratio/high_mean": 0.10793161019682884, "clip_ratio/high_max": 0.10793161019682884, "clip_ratio/region_mean": 0.2000612411648035, "reward_total_mean": 0.4768216609954834, "reward_meter_mean": 0.7325787544250488, "reward_meter_std": 0.39579838514328003, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.961837112903595, "reward_repeat_soft_std": 0.0018749026348814368, "reward_judge_quality_mean": 0.36374998092651367, "reward_judge_quality_std": 0.13265825808048248, "reward_total_composite_mean": 0.4768216609954834, "reward_total_composite_std": 0.21741239726543427} {"timestamp_utc": "2026-04-13T07:45:02Z", "mode": "train", "global_step": 169, "epoch": 0.016976393771973883, "loss": 0.0113, "grad_norm": 21.640043258666992, "learning_rate": 9.490909090909092e-06, "num_tokens": 312307.0, "completions/mean_length": 38.875, "completions/min_length": 32.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.875, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.7079165577888489, "rewards/meter/std": 0.40104302763938904, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9991671442985535, "rewards/repeat_soft/std": 0.0023556507658213377, "rewards/judge_quality/mean": 0.6937500238418579, "rewards/judge_quality/std": 0.2728651165962219, "rewards/total_composite/mean": 0.6812629699707031, "rewards/total_composite/std": 0.23844361305236816, "reward": 0.6812629699707031, "reward_std": 0.23844359815120697, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22272655367851257, "sampling/sampling_logp_difference/max": 1.4653444290161133, "sampling/importance_sampling_ratio/min": 0.2309984266757965, "sampling/importance_sampling_ratio/mean": 1.0448509454727173, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.164816826581955, "clip_ratio/low_mean": 0.0858415924012661, "clip_ratio/low_min": 0.0858415924012661, "clip_ratio/high_mean": 0.09395995084196329, "clip_ratio/high_max": 0.09395995084196329, "clip_ratio/region_mean": 0.1798015432432294, "reward_total_mean": 0.6812629699707031, "reward_meter_mean": 0.7079165577888489, "reward_meter_std": 0.40104302763938904, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9991671442985535, "reward_repeat_soft_std": 0.0023556507658213377, "reward_judge_quality_mean": 0.6937500238418579, "reward_judge_quality_std": 0.2728651165962219, "reward_total_composite_mean": 0.6812629699707031, "reward_total_composite_std": 0.23844361305236816} {"timestamp_utc": "2026-04-13T07:45:09Z", "mode": "train", "global_step": 170, "epoch": 0.017076845806127575, "loss": 0.0214, "grad_norm": 14.737093925476074, "learning_rate": 9.487878787878788e-06, "num_tokens": 313806.0, "completions/mean_length": 40.375, "completions/min_length": 35.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.375, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.8741497993469238, "rewards/meter/std": 0.3062879741191864, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9916309118270874, "rewards/repeat_soft/std": 0.010093681514263153, "rewards/judge_quality/mean": 0.5900000333786011, "rewards/judge_quality/std": 0.22696760296821594, "rewards/total_composite/mean": 0.6944446563720703, "rewards/total_composite/std": 0.19079254567623138, "reward": 0.6944446563720703, "reward_std": 0.1907925307750702, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21184924244880676, "sampling/sampling_logp_difference/max": 1.7176275253295898, "sampling/importance_sampling_ratio/min": 0.17949149012565613, "sampling/importance_sampling_ratio/mean": 1.075750470161438, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6572673618793488, "clip_ratio/low_mean": 0.15218430012464523, "clip_ratio/low_min": 0.15218430012464523, "clip_ratio/high_mean": 0.07383041083812714, "clip_ratio/high_max": 0.07383041083812714, "clip_ratio/region_mean": 0.22601471096277237, "reward_total_mean": 0.6944446563720703, "reward_meter_mean": 0.8741497993469238, "reward_meter_std": 0.3062879741191864, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9916309118270874, "reward_repeat_soft_std": 0.010093681514263153, "reward_judge_quality_mean": 0.5900000333786011, "reward_judge_quality_std": 0.22696760296821594, "reward_total_composite_mean": 0.6944446563720703, "reward_total_composite_std": 0.19079254567623138} {"timestamp_utc": "2026-04-13T07:45:16Z", "mode": "train", "global_step": 171, "epoch": 0.017177297840281266, "loss": 0.1001, "grad_norm": 15.172825813293457, "learning_rate": 9.484848484848485e-06, "num_tokens": 315546.0, "completions/mean_length": 51.5, "completions/min_length": 32.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.5, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.7064719200134277, "rewards/meter/std": 0.3983866274356842, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9972226023674011, "rewards/repeat_soft/std": 0.007504946552217007, "rewards/judge_quality/mean": 0.6187499761581421, "rewards/judge_quality/std": 0.24976776540279388, "rewards/total_composite/mean": 0.6153678297996521, "rewards/total_composite/std": 0.17465214431285858, "reward": 0.6153678297996521, "reward_std": 0.17465214431285858, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2158244103193283, "sampling/sampling_logp_difference/max": 1.5815305709838867, "sampling/importance_sampling_ratio/min": 0.20566007494926453, "sampling/importance_sampling_ratio/mean": 1.0184203386306763, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.242932051420212, "clip_ratio/low_mean": 0.0644509494304657, "clip_ratio/low_min": 0.0644509494304657, "clip_ratio/high_mean": 0.1464561503380537, "clip_ratio/high_max": 0.1464561503380537, "clip_ratio/region_mean": 0.2109070997685194, "reward_total_mean": 0.6153678297996521, "reward_meter_mean": 0.7064719200134277, "reward_meter_std": 0.3983866274356842, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9972226023674011, "reward_repeat_soft_std": 0.007504946552217007, "reward_judge_quality_mean": 0.6187499761581421, "reward_judge_quality_std": 0.24976776540279388, "reward_total_composite_mean": 0.6153678297996521, "reward_total_composite_std": 0.17465214431285858} {"timestamp_utc": "2026-04-13T07:45:23Z", "mode": "train", "global_step": 172, "epoch": 0.017277749874434957, "loss": -0.1089, "grad_norm": 19.612377166748047, "learning_rate": 9.481818181818182e-06, "num_tokens": 317117.0, "completions/mean_length": 33.375, "completions/min_length": 24.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.375, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.5997594594955444, "rewards/meter/std": 0.3780299425125122, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9781632423400879, "rewards/repeat_soft/std": 0.02724776230752468, "rewards/judge_quality/mean": 0.6150000095367432, "rewards/judge_quality/std": 0.25286927819252014, "rewards/total_composite/mean": 0.5745712518692017, "rewards/total_composite/std": 0.17878694832324982, "reward": 0.5745712518692017, "reward_std": 0.17878693342208862, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2273896038532257, "sampling/sampling_logp_difference/max": 1.4090204238891602, "sampling/importance_sampling_ratio/min": 0.2443825751543045, "sampling/importance_sampling_ratio/mean": 1.0568751096725464, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2210111021995544, "clip_ratio/low_mean": 0.07622195687144995, "clip_ratio/low_min": 0.07622195687144995, "clip_ratio/high_mean": 0.11674907617270947, "clip_ratio/high_max": 0.11674907617270947, "clip_ratio/region_mean": 0.1929710330441594, "reward_total_mean": 0.5745712518692017, "reward_meter_mean": 0.5997594594955444, "reward_meter_std": 0.3780299425125122, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9781632423400879, "reward_repeat_soft_std": 0.02724776230752468, "reward_judge_quality_mean": 0.6150000095367432, "reward_judge_quality_std": 0.25286927819252014, "reward_total_composite_mean": 0.5745712518692017, "reward_total_composite_std": 0.17878694832324982} {"timestamp_utc": "2026-04-13T07:45:30Z", "mode": "train", "global_step": 173, "epoch": 0.017378201908588648, "loss": -0.0445, "grad_norm": 16.774700164794922, "learning_rate": 9.47878787878788e-06, "num_tokens": 318678.0, "completions/mean_length": 31.125, "completions/min_length": 24.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.125, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.5051141381263733, "rewards/meter/std": 0.3768838942050934, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.951198160648346, "rewards/repeat_soft/std": 0.023612266406416893, "rewards/judge_quality/mean": 0.7275000214576721, "rewards/judge_quality/std": 0.3564407229423523, "rewards/total_composite/mean": 0.5817102193832397, "rewards/total_composite/std": 0.2026408612728119, "reward": 0.5817102193832397, "reward_std": 0.20264087617397308, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19832772016525269, "sampling/sampling_logp_difference/max": 1.467167854309082, "sampling/importance_sampling_ratio/min": 0.23057760298252106, "sampling/importance_sampling_ratio/mean": 1.015305519104004, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5307548269629478, "clip_ratio/low_mean": 0.08864182699471712, "clip_ratio/low_min": 0.08864182699471712, "clip_ratio/high_mean": 0.11992127541452646, "clip_ratio/high_max": 0.11992127541452646, "clip_ratio/region_mean": 0.20856310240924358, "reward_total_mean": 0.5817102193832397, "reward_meter_mean": 0.5051141381263733, "reward_meter_std": 0.3768838942050934, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.951198160648346, "reward_repeat_soft_std": 0.023612266406416893, "reward_judge_quality_mean": 0.7275000214576721, "reward_judge_quality_std": 0.3564407229423523, "reward_total_composite_mean": 0.5817102193832397, "reward_total_composite_std": 0.2026408612728119} {"timestamp_utc": "2026-04-13T07:45:37Z", "mode": "train", "global_step": 174, "epoch": 0.01747865394274234, "loss": 0.0959, "grad_norm": 10.710179328918457, "learning_rate": 9.475757575757577e-06, "num_tokens": 320342.0, "completions/mean_length": 51.0, "completions/min_length": 38.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.0, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.15989206731319427, "rewards/meter/std": 0.31738677620887756, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9953173398971558, "rewards/repeat_soft/std": 0.007823141291737556, "rewards/judge_quality/mean": 0.4099999964237213, "rewards/judge_quality/std": 0.06633248925209045, "rewards/total_composite/mean": 0.39320605993270874, "rewards/total_composite/std": 0.08556443452835083, "reward": 0.39320605993270874, "reward_std": 0.08556442707777023, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24069498479366302, "sampling/sampling_logp_difference/max": 1.5249691009521484, "sampling/importance_sampling_ratio/min": 0.21762779355049133, "sampling/importance_sampling_ratio/mean": 1.0592713356018066, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.0124190747737885, "clip_ratio/low_mean": 0.15173242706805468, "clip_ratio/low_min": 0.15173242706805468, "clip_ratio/high_mean": 0.037790697067976, "clip_ratio/high_max": 0.037790697067976, "clip_ratio/region_mean": 0.18952312413603067, "reward_total_mean": 0.39320605993270874, "reward_meter_mean": 0.15989206731319427, "reward_meter_std": 0.31738677620887756, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9953173398971558, "reward_repeat_soft_std": 0.007823141291737556, "reward_judge_quality_mean": 0.4099999964237213, "reward_judge_quality_std": 0.06633248925209045, "reward_total_composite_mean": 0.39320605993270874, "reward_total_composite_std": 0.08556443452835083} {"timestamp_utc": "2026-04-13T07:45:44Z", "mode": "train", "global_step": 175, "epoch": 0.01757910597689603, "loss": 0.1424, "grad_norm": 16.74935531616211, "learning_rate": 9.472727272727274e-06, "num_tokens": 321934.0, "completions/mean_length": 44.0, "completions/min_length": 33.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.0, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.8714777827262878, "rewards/meter/std": 0.27093425393104553, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9930651187896729, "rewards/repeat_soft/std": 0.00821753591299057, "rewards/judge_quality/mean": 0.6775000095367432, "rewards/judge_quality/std": 0.17210878431797028, "rewards/total_composite/mean": 0.7505009174346924, "rewards/total_composite/std": 0.1681082397699356, "reward": 0.7505009174346924, "reward_std": 0.16810822486877441, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16729728877544403, "sampling/sampling_logp_difference/max": 1.657027244567871, "sampling/importance_sampling_ratio/min": 0.1907050460577011, "sampling/importance_sampling_ratio/mean": 1.0261846780776978, "sampling/importance_sampling_ratio/max": 1.8706589937210083, "entropy": 1.6106469556689262, "clip_ratio/low_mean": 0.039352625608444214, "clip_ratio/low_min": 0.039352625608444214, "clip_ratio/high_mean": 0.11767790373414755, "clip_ratio/high_max": 0.11767790373414755, "clip_ratio/region_mean": 0.15703052934259176, "reward_total_mean": 0.7505009174346924, "reward_meter_mean": 0.8714777827262878, "reward_meter_std": 0.27093425393104553, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9930651187896729, "reward_repeat_soft_std": 0.00821753591299057, "reward_judge_quality_mean": 0.6775000095367432, "reward_judge_quality_std": 0.17210878431797028, "reward_total_composite_mean": 0.7505009174346924, "reward_total_composite_std": 0.1681082397699356} {"timestamp_utc": "2026-04-13T07:45:51Z", "mode": "train", "global_step": 176, "epoch": 0.017679558011049725, "loss": 0.067, "grad_norm": 23.75896453857422, "learning_rate": 9.469696969696971e-06, "num_tokens": 323519.0, "completions/mean_length": 40.125, "completions/min_length": 34.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.125, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.5985807180404663, "rewards/meter/std": 0.43099549412727356, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.992364764213562, "rewards/repeat_soft/std": 0.007917179726064205, "rewards/judge_quality/mean": 0.6150000095367432, "rewards/judge_quality/std": 0.25286927819252014, "rewards/total_composite/mean": 0.587218165397644, "rewards/total_composite/std": 0.3418886363506317, "reward": 0.587218165397644, "reward_std": 0.3418886363506317, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21122638881206512, "sampling/sampling_logp_difference/max": 2.2637810707092285, "sampling/importance_sampling_ratio/min": 0.1039566695690155, "sampling/importance_sampling_ratio/mean": 1.0143921375274658, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0883154422044754, "clip_ratio/low_mean": 0.122177729383111, "clip_ratio/low_min": 0.122177729383111, "clip_ratio/high_mean": 0.07360660284757614, "clip_ratio/high_max": 0.07360660284757614, "clip_ratio/region_mean": 0.19578433223068714, "reward_total_mean": 0.587218165397644, "reward_meter_mean": 0.5985807180404663, "reward_meter_std": 0.43099549412727356, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.992364764213562, "reward_repeat_soft_std": 0.007917179726064205, "reward_judge_quality_mean": 0.6150000095367432, "reward_judge_quality_std": 0.25286927819252014, "reward_total_composite_mean": 0.587218165397644, "reward_total_composite_std": 0.3418886363506317} {"timestamp_utc": "2026-04-13T07:45:59Z", "mode": "train", "global_step": 177, "epoch": 0.017780010045203416, "loss": 0.0198, "grad_norm": 13.65584659576416, "learning_rate": 9.466666666666667e-06, "num_tokens": 325161.0, "completions/mean_length": 54.25, "completions/min_length": 36.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.25, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.3896534740924835, "rewards/meter/std": 0.318272203207016, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9989882707595825, "rewards/repeat_soft/std": 0.002861686749383807, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.481770396232605, "rewards/total_composite/std": 0.10922771692276001, "reward": 0.481770396232605, "reward_std": 0.10922770202159882, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22601468861103058, "sampling/sampling_logp_difference/max": 1.635115623474121, "sampling/importance_sampling_ratio/min": 0.1949298232793808, "sampling/importance_sampling_ratio/mean": 1.0414087772369385, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6458567082881927, "clip_ratio/low_mean": 0.0945226140320301, "clip_ratio/low_min": 0.0945226140320301, "clip_ratio/high_mean": 0.08183415047824383, "clip_ratio/high_max": 0.08183415047824383, "clip_ratio/region_mean": 0.17635676451027393, "reward_total_mean": 0.481770396232605, "reward_meter_mean": 0.3896534740924835, "reward_meter_std": 0.318272203207016, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9989882707595825, "reward_repeat_soft_std": 0.002861686749383807, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.481770396232605, "reward_total_composite_std": 0.10922771692276001} {"timestamp_utc": "2026-04-13T07:46:07Z", "mode": "train", "global_step": 178, "epoch": 0.017880462079357107, "loss": 0.0195, "grad_norm": 8.766780853271484, "learning_rate": 9.463636363636364e-06, "num_tokens": 327668.0, "completions/mean_length": 115.375, "completions/min_length": 95.0, "completions/max_length": 161.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.375, "completions/min_terminated_length": 95.0, "completions/max_terminated_length": 161.0, "rewards/meter/mean": 0.7110608816146851, "rewards/meter/std": 0.3456106185913086, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.1035098284482956, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9969117641448975, "rewards/repeat_soft/std": 0.0029427569825202227, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.43055757880210876, "rewards/total_composite/std": 0.20038874447345734, "reward": 0.43055757880210876, "reward_std": 0.20038872957229614, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2415567934513092, "sampling/sampling_logp_difference/max": 1.5120110511779785, "sampling/importance_sampling_ratio/min": 0.22046616673469543, "sampling/importance_sampling_ratio/mean": 1.0475821495056152, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.4164766371250153, "clip_ratio/low_mean": 0.07863680832087994, "clip_ratio/low_min": 0.07863680832087994, "clip_ratio/high_mean": 0.1258973926305771, "clip_ratio/high_max": 0.1258973926305771, "clip_ratio/region_mean": 0.20453420095145702, "reward_total_mean": 0.43055757880210876, "reward_meter_mean": 0.7110608816146851, "reward_meter_std": 0.3456106185913086, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.1035098284482956, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9969117641448975, "reward_repeat_soft_std": 0.0029427569825202227, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.43055757880210876, "reward_total_composite_std": 0.20038874447345734} {"timestamp_utc": "2026-04-13T07:46:14Z", "mode": "train", "global_step": 179, "epoch": 0.0179809141135108, "loss": 0.0365, "grad_norm": 11.071876525878906, "learning_rate": 9.460606060606061e-06, "num_tokens": 329759.0, "completions/mean_length": 85.375, "completions/min_length": 73.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.375, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.4272414445877075, "rewards/meter/std": 0.25779473781585693, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.996740996837616, "rewards/repeat_soft/std": 0.0038489243015646935, "rewards/judge_quality/mean": 0.41499999165534973, "rewards/judge_quality/std": 0.14520922303199768, "rewards/total_composite/mean": 0.47163498401641846, "rewards/total_composite/std": 0.0906149297952652, "reward": 0.47163498401641846, "reward_std": 0.0906149372458458, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24609750509262085, "sampling/sampling_logp_difference/max": 1.5515518188476562, "sampling/importance_sampling_ratio/min": 0.21191886067390442, "sampling/importance_sampling_ratio/mean": 1.0612713098526, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.618737071752548, "clip_ratio/low_mean": 0.10393072944134474, "clip_ratio/low_min": 0.10393072944134474, "clip_ratio/high_mean": 0.07959592528641224, "clip_ratio/high_max": 0.07959592528641224, "clip_ratio/region_mean": 0.18352665472775698, "reward_total_mean": 0.47163498401641846, "reward_meter_mean": 0.4272414445877075, "reward_meter_std": 0.25779473781585693, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.996740996837616, "reward_repeat_soft_std": 0.0038489243015646935, "reward_judge_quality_mean": 0.41499999165534973, "reward_judge_quality_std": 0.14520922303199768, "reward_total_composite_mean": 0.47163498401641846, "reward_total_composite_std": 0.0906149297952652} {"timestamp_utc": "2026-04-13T07:46:22Z", "mode": "train", "global_step": 180, "epoch": 0.01808136614766449, "loss": 0.0235, "grad_norm": 11.487159729003906, "learning_rate": 9.457575757575759e-06, "num_tokens": 331979.0, "completions/mean_length": 79.5, "completions/min_length": 61.0, "completions/max_length": 125.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.5, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.31564873456954956, "rewards/meter/std": 0.30156204104423523, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9972239136695862, "rewards/repeat_soft/std": 0.0030039141420274973, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.414653480052948, "rewards/total_composite/std": 0.08931313455104828, "reward": 0.414653480052948, "reward_std": 0.08931312710046768, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2503102719783783, "sampling/sampling_logp_difference/max": 1.6738219261169434, "sampling/importance_sampling_ratio/min": 0.18752898275852203, "sampling/importance_sampling_ratio/mean": 1.042333960533142, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.0768066942691803, "clip_ratio/low_mean": 0.13392947241663933, "clip_ratio/low_min": 0.13392947241663933, "clip_ratio/high_mean": 0.0785590372979641, "clip_ratio/high_max": 0.0785590372979641, "clip_ratio/region_mean": 0.21248850971460342, "reward_total_mean": 0.414653480052948, "reward_meter_mean": 0.31564873456954956, "reward_meter_std": 0.30156204104423523, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9972239136695862, "reward_repeat_soft_std": 0.0030039141420274973, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.414653480052948, "reward_total_composite_std": 0.08931313455104828} {"timestamp_utc": "2026-04-13T07:46:30Z", "mode": "train", "global_step": 181, "epoch": 0.01818181818181818, "loss": 0.0879, "grad_norm": 12.22265338897705, "learning_rate": 9.454545454545456e-06, "num_tokens": 333734.0, "completions/mean_length": 59.375, "completions/min_length": 46.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.375, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.4035565257072449, "rewards/meter/std": 0.4075183570384979, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9969116449356079, "rewards/repeat_soft/std": 0.0037362936418503523, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.2327437400817871, "rewards/total_composite/mean": 0.48305100202560425, "rewards/total_composite/std": 0.20548434555530548, "reward": 0.48305100202560425, "reward_std": 0.20548434555530548, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2156466543674469, "sampling/sampling_logp_difference/max": 1.5678300857543945, "sampling/importance_sampling_ratio/min": 0.20849710702896118, "sampling/importance_sampling_ratio/mean": 1.0440673828125, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.9228354692459106, "clip_ratio/low_mean": 0.1170395202934742, "clip_ratio/low_min": 0.1170395202934742, "clip_ratio/high_mean": 0.06735390983521938, "clip_ratio/high_max": 0.06735390983521938, "clip_ratio/region_mean": 0.18439343012869358, "reward_total_mean": 0.48305100202560425, "reward_meter_mean": 0.4035565257072449, "reward_meter_std": 0.4075183570384979, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9969116449356079, "reward_repeat_soft_std": 0.0037362936418503523, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.2327437400817871, "reward_total_composite_mean": 0.48305100202560425, "reward_total_composite_std": 0.20548434555530548} {"timestamp_utc": "2026-04-13T07:46:36Z", "mode": "train", "global_step": 182, "epoch": 0.018282270215971872, "loss": 0.2062, "grad_norm": 26.969940185546875, "learning_rate": 9.451515151515153e-06, "num_tokens": 335325.0, "completions/mean_length": 34.875, "completions/min_length": 24.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.875, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.5274512767791748, "rewards/meter/std": 0.4633592963218689, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.977503776550293, "rewards/repeat_soft/std": 0.020300719887018204, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.5399748086929321, "rewards/total_composite/std": 0.1919652372598648, "reward": 0.5399748086929321, "reward_std": 0.1919652372598648, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23039452731609344, "sampling/sampling_logp_difference/max": 2.461104393005371, "sampling/importance_sampling_ratio/min": 0.08534064888954163, "sampling/importance_sampling_ratio/mean": 0.9769449234008789, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1152411550283432, "clip_ratio/low_mean": 0.07630632631480694, "clip_ratio/low_min": 0.07630632631480694, "clip_ratio/high_mean": 0.14935007132589817, "clip_ratio/high_max": 0.14935007132589817, "clip_ratio/region_mean": 0.2256563976407051, "reward_total_mean": 0.5399748086929321, "reward_meter_mean": 0.5274512767791748, "reward_meter_std": 0.4633592963218689, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.977503776550293, "reward_repeat_soft_std": 0.020300719887018204, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.5399748086929321, "reward_total_composite_std": 0.1919652372598648} {"timestamp_utc": "2026-04-13T07:46:42Z", "mode": "train", "global_step": 183, "epoch": 0.018382722250125567, "loss": 0.047, "grad_norm": 23.409812927246094, "learning_rate": 9.448484848484849e-06, "num_tokens": 336977.0, "completions/mean_length": 33.5, "completions/min_length": 30.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.5, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.8624000549316406, "rewards/meter/std": 0.3189335763454437, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9998470544815063, "rewards/repeat_soft/std": 0.00043267954606562853, "rewards/judge_quality/mean": 0.6850000023841858, "rewards/judge_quality/std": 0.2512255907058716, "rewards/total_composite/mean": 0.7162233591079712, "rewards/total_composite/std": 0.19540704786777496, "reward": 0.7162233591079712, "reward_std": 0.19540704786777496, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14572322368621826, "sampling/sampling_logp_difference/max": 1.710498332977295, "sampling/importance_sampling_ratio/min": 0.1807756870985031, "sampling/importance_sampling_ratio/mean": 1.0187908411026, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.661162231117487, "clip_ratio/low_mean": 0.06761159980669618, "clip_ratio/low_min": 0.06761159980669618, "clip_ratio/high_mean": 0.0531698577105999, "clip_ratio/high_max": 0.0531698577105999, "clip_ratio/region_mean": 0.12078145751729608, "reward_total_mean": 0.7162233591079712, "reward_meter_mean": 0.8624000549316406, "reward_meter_std": 0.3189335763454437, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9998470544815063, "reward_repeat_soft_std": 0.00043267954606562853, "reward_judge_quality_mean": 0.6850000023841858, "reward_judge_quality_std": 0.2512255907058716, "reward_total_composite_mean": 0.7162233591079712, "reward_total_composite_std": 0.19540704786777496} {"timestamp_utc": "2026-04-13T07:46:49Z", "mode": "train", "global_step": 184, "epoch": 0.018483174284279258, "loss": 0.0993, "grad_norm": 15.767330169677734, "learning_rate": 9.445454545454546e-06, "num_tokens": 338608.0, "completions/mean_length": 47.875, "completions/min_length": 37.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.875, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.5631375908851624, "rewards/meter/std": 0.3976224958896637, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9984055757522583, "rewards/repeat_soft/std": 0.0025164084509015083, "rewards/judge_quality/mean": 0.47999998927116394, "rewards/judge_quality/std": 0.09754122048616409, "rewards/total_composite/mean": 0.5147405862808228, "rewards/total_composite/std": 0.11590421199798584, "reward": 0.5147405862808228, "reward_std": 0.11590423434972763, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22204120457172394, "sampling/sampling_logp_difference/max": 1.1929655075073242, "sampling/importance_sampling_ratio/min": 0.30332043766975403, "sampling/importance_sampling_ratio/mean": 1.0571694374084473, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.678367853164673, "clip_ratio/low_mean": 0.0853744950145483, "clip_ratio/low_min": 0.0853744950145483, "clip_ratio/high_mean": 0.12821896374225616, "clip_ratio/high_max": 0.12821896374225616, "clip_ratio/region_mean": 0.21359345875680447, "reward_total_mean": 0.5147405862808228, "reward_meter_mean": 0.5631375908851624, "reward_meter_std": 0.3976224958896637, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9984055757522583, "reward_repeat_soft_std": 0.0025164084509015083, "reward_judge_quality_mean": 0.47999998927116394, "reward_judge_quality_std": 0.09754122048616409, "reward_total_composite_mean": 0.5147405862808228, "reward_total_composite_std": 0.11590421199798584} {"timestamp_utc": "2026-04-13T07:46:56Z", "mode": "train", "global_step": 185, "epoch": 0.01858362631843295, "loss": -0.0371, "grad_norm": 17.47907257080078, "learning_rate": 9.442424242424243e-06, "num_tokens": 340039.0, "completions/mean_length": 26.875, "completions/min_length": 19.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.875, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.549547553062439, "rewards/meter/std": 0.46725091338157654, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9615821838378906, "rewards/repeat_soft/std": 0.002596014179289341, "rewards/judge_quality/mean": 0.48625001311302185, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.5350043773651123, "rewards/total_composite/std": 0.19496038556098938, "reward": 0.5350043773651123, "reward_std": 0.19496038556098938, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21452338993549347, "sampling/sampling_logp_difference/max": 1.4326086044311523, "sampling/importance_sampling_ratio/min": 0.2386854887008667, "sampling/importance_sampling_ratio/mean": 1.0544756650924683, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4443026930093765, "clip_ratio/low_mean": 0.11357326246798038, "clip_ratio/low_min": 0.11357326246798038, "clip_ratio/high_mean": 0.12992424704134464, "clip_ratio/high_max": 0.12992424704134464, "clip_ratio/region_mean": 0.24349750950932503, "reward_total_mean": 0.5350043773651123, "reward_meter_mean": 0.549547553062439, "reward_meter_std": 0.46725091338157654, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9615821838378906, "reward_repeat_soft_std": 0.002596014179289341, "reward_judge_quality_mean": 0.48625001311302185, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.5350043773651123, "reward_total_composite_std": 0.19496038556098938} {"timestamp_utc": "2026-04-13T07:47:02Z", "mode": "train", "global_step": 186, "epoch": 0.01868407835258664, "loss": 0.0235, "grad_norm": 17.311351776123047, "learning_rate": 9.43939393939394e-06, "num_tokens": 342081.0, "completions/mean_length": 72.25, "completions/min_length": 54.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.25, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.3997536301612854, "rewards/meter/std": 0.37315547466278076, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9957786798477173, "rewards/repeat_soft/std": 0.005814037751406431, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.398880273103714, "rewards/total_composite/std": 0.1924026906490326, "reward": 0.398880273103714, "reward_std": 0.1924026757478714, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.262917697429657, "sampling/sampling_logp_difference/max": 3.0831079483032227, "sampling/importance_sampling_ratio/min": 0.04581664130091667, "sampling/importance_sampling_ratio/mean": 1.00346040725708, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.003795847296715, "clip_ratio/low_mean": 0.12207727693021297, "clip_ratio/low_min": 0.12207727693021297, "clip_ratio/high_mean": 0.082015885040164, "clip_ratio/high_max": 0.082015885040164, "clip_ratio/region_mean": 0.20409316197037697, "reward_total_mean": 0.398880273103714, "reward_meter_mean": 0.3997536301612854, "reward_meter_std": 0.37315547466278076, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9957786798477173, "reward_repeat_soft_std": 0.005814037751406431, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.398880273103714, "reward_total_composite_std": 0.1924026906490326} {"timestamp_utc": "2026-04-13T07:47:11Z", "mode": "train", "global_step": 187, "epoch": 0.01878453038674033, "loss": 0.0127, "grad_norm": 7.8910722732543945, "learning_rate": 9.436363636363636e-06, "num_tokens": 344835.0, "completions/mean_length": 146.25, "completions/min_length": 104.0, "completions/max_length": 181.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 146.25, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 181.0, "rewards/meter/mean": 0.5559006929397583, "rewards/meter/std": 0.28145334124565125, "rewards/count_adherence/mean": 0.8958333134651184, "rewards/count_adherence/std": 0.08625820279121399, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9929571151733398, "rewards/repeat_soft/std": 0.007836537435650826, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.4760981798171997, "rewards/total_composite/std": 0.09441790729761124, "reward": 0.4760981798171997, "reward_std": 0.09441790729761124, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22341546416282654, "sampling/sampling_logp_difference/max": 1.3534693717956543, "sampling/importance_sampling_ratio/min": 0.2583424150943756, "sampling/importance_sampling_ratio/mean": 1.062045693397522, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.1189622282981873, "clip_ratio/low_mean": 0.0684958714991808, "clip_ratio/low_min": 0.0684958714991808, "clip_ratio/high_mean": 0.1289503499865532, "clip_ratio/high_max": 0.1289503499865532, "clip_ratio/region_mean": 0.19744622148573399, "reward_total_mean": 0.4760981798171997, "reward_meter_mean": 0.5559006929397583, "reward_meter_std": 0.28145334124565125, "reward_count_adherence_mean": 0.8958333134651184, "reward_count_adherence_std": 0.08625820279121399, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9929571151733398, "reward_repeat_soft_std": 0.007836537435650826, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.4760981798171997, "reward_total_composite_std": 0.09441790729761124} {"timestamp_utc": "2026-04-13T07:47:17Z", "mode": "train", "global_step": 188, "epoch": 0.018884982420894023, "loss": -0.1187, "grad_norm": 15.400172233581543, "learning_rate": 9.433333333333335e-06, "num_tokens": 346495.0, "completions/mean_length": 50.5, "completions/min_length": 39.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.5, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.8723356127738953, "rewards/meter/std": 0.2369154691696167, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9981918334960938, "rewards/repeat_soft/std": 0.004290272947400808, "rewards/judge_quality/mean": 0.48625001311302185, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.6304661631584167, "rewards/total_composite/std": 0.14154669642448425, "reward": 0.6304661631584167, "reward_std": 0.14154671132564545, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23977959156036377, "sampling/sampling_logp_difference/max": 1.277449607849121, "sampling/importance_sampling_ratio/min": 0.2787473201751709, "sampling/importance_sampling_ratio/mean": 1.0627295970916748, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.198775589466095, "clip_ratio/low_mean": 0.1782637145370245, "clip_ratio/low_min": 0.1782637145370245, "clip_ratio/high_mean": 0.045219458639621735, "clip_ratio/high_max": 0.045219458639621735, "clip_ratio/region_mean": 0.22348317317664623, "reward_total_mean": 0.6304661631584167, "reward_meter_mean": 0.8723356127738953, "reward_meter_std": 0.2369154691696167, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9981918334960938, "reward_repeat_soft_std": 0.004290272947400808, "reward_judge_quality_mean": 0.48625001311302185, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.6304661631584167, "reward_total_composite_std": 0.14154669642448425} {"timestamp_utc": "2026-04-13T07:47:24Z", "mode": "train", "global_step": 189, "epoch": 0.018985434455047714, "loss": 0.1108, "grad_norm": 35.58707809448242, "learning_rate": 9.43030303030303e-06, "num_tokens": 348008.0, "completions/mean_length": 33.125, "completions/min_length": 29.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.125, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.6826807260513306, "rewards/meter/std": 0.306144118309021, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9814399480819702, "rewards/repeat_soft/std": 0.027393506839871407, "rewards/judge_quality/mean": 0.7950000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.6602128744125366, "rewards/total_composite/std": 0.3298247158527374, "reward": 0.6602128744125366, "reward_std": 0.32982468605041504, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23146715760231018, "sampling/sampling_logp_difference/max": 2.0614590644836426, "sampling/importance_sampling_ratio/min": 0.12726813554763794, "sampling/importance_sampling_ratio/mean": 1.003521203994751, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.97225097194314, "clip_ratio/low_mean": 0.07323569525033236, "clip_ratio/low_min": 0.07323569525033236, "clip_ratio/high_mean": 0.10484400764107704, "clip_ratio/high_max": 0.10484400764107704, "clip_ratio/region_mean": 0.1780797028914094, "reward_total_mean": 0.6602128744125366, "reward_meter_mean": 0.6826807260513306, "reward_meter_std": 0.306144118309021, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9814399480819702, "reward_repeat_soft_std": 0.027393506839871407, "reward_judge_quality_mean": 0.7950000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.6602128744125366, "reward_total_composite_std": 0.3298247158527374} {"timestamp_utc": "2026-04-13T07:47:32Z", "mode": "train", "global_step": 190, "epoch": 0.019085886489201405, "loss": 0.068, "grad_norm": 20.739145278930664, "learning_rate": 9.427272727272728e-06, "num_tokens": 349685.0, "completions/mean_length": 40.625, "completions/min_length": 16.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.625, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.3190385401248932, "rewards/meter/std": 0.38571831583976746, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9899997115135193, "rewards/repeat_soft/std": 0.01640298217535019, "rewards/judge_quality/mean": 0.45124998688697815, "rewards/judge_quality/std": 0.13767845928668976, "rewards/total_composite/mean": 0.38480043411254883, "rewards/total_composite/std": 0.1965453326702118, "reward": 0.38480043411254883, "reward_std": 0.1965453326702118, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.25145721435546875, "sampling/sampling_logp_difference/max": 1.5635175704956055, "sampling/importance_sampling_ratio/min": 0.20939819514751434, "sampling/importance_sampling_ratio/mean": 1.0337738990783691, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.717229127883911, "clip_ratio/low_mean": 0.11086361482739449, "clip_ratio/low_min": 0.11086361482739449, "clip_ratio/high_mean": 0.1254812590777874, "clip_ratio/high_max": 0.1254812590777874, "clip_ratio/region_mean": 0.23634487390518188, "reward_total_mean": 0.38480043411254883, "reward_meter_mean": 0.3190385401248932, "reward_meter_std": 0.38571831583976746, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9899997115135193, "reward_repeat_soft_std": 0.01640298217535019, "reward_judge_quality_mean": 0.45124998688697815, "reward_judge_quality_std": 0.13767845928668976, "reward_total_composite_mean": 0.38480043411254883, "reward_total_composite_std": 0.1965453326702118} {"timestamp_utc": "2026-04-13T07:47:42Z", "mode": "train", "global_step": 191, "epoch": 0.0191863385233551, "loss": 0.0167, "grad_norm": 7.381566524505615, "learning_rate": 9.424242424242425e-06, "num_tokens": 352401.0, "completions/mean_length": 154.5, "completions/min_length": 111.0, "completions/max_length": 185.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 154.5, "completions/min_terminated_length": 111.0, "completions/max_terminated_length": 185.0, "rewards/meter/mean": 0.8647301197052002, "rewards/meter/std": 0.28420892357826233, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.1035098284482956, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9960473775863647, "rewards/repeat_soft/std": 0.004961573984473944, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.48406609892845154, "rewards/total_composite/std": 0.21130195260047913, "reward": 0.48406609892845154, "reward_std": 0.21130193769931793, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20887885987758636, "sampling/sampling_logp_difference/max": 2.524847984313965, "sampling/importance_sampling_ratio/min": 0.08007048070430756, "sampling/importance_sampling_ratio/mean": 1.0513540506362915, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.996433287858963, "clip_ratio/low_mean": 0.0464102178812027, "clip_ratio/low_min": 0.0464102178812027, "clip_ratio/high_mean": 0.1237588468939066, "clip_ratio/high_max": 0.1237588468939066, "clip_ratio/region_mean": 0.1701690647751093, "reward_total_mean": 0.48406609892845154, "reward_meter_mean": 0.8647301197052002, "reward_meter_std": 0.28420892357826233, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.1035098284482956, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9960473775863647, "reward_repeat_soft_std": 0.004961573984473944, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.48406609892845154, "reward_total_composite_std": 0.21130195260047913} {"timestamp_utc": "2026-04-13T07:47:49Z", "mode": "train", "global_step": 192, "epoch": 0.01928679055750879, "loss": -0.0564, "grad_norm": 21.528614044189453, "learning_rate": 9.421212121212122e-06, "num_tokens": 354050.0, "completions/mean_length": 35.125, "completions/min_length": 22.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.125, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.4956907033920288, "rewards/meter/std": 0.35053014755249023, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9955316781997681, "rewards/repeat_soft/std": 0.006094147451221943, "rewards/judge_quality/mean": 0.4312499761581421, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.48836007714271545, "rewards/total_composite/std": 0.09807529300451279, "reward": 0.48836007714271545, "reward_std": 0.09807528555393219, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2227940410375595, "sampling/sampling_logp_difference/max": 1.2411231994628906, "sampling/importance_sampling_ratio/min": 0.2890593707561493, "sampling/importance_sampling_ratio/mean": 1.0087708234786987, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9255905747413635, "clip_ratio/low_mean": 0.12833823263645172, "clip_ratio/low_min": 0.12833823263645172, "clip_ratio/high_mean": 0.08295800909399986, "clip_ratio/high_max": 0.08295800909399986, "clip_ratio/region_mean": 0.21129624173045158, "reward_total_mean": 0.48836007714271545, "reward_meter_mean": 0.4956907033920288, "reward_meter_std": 0.35053014755249023, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9955316781997681, "reward_repeat_soft_std": 0.006094147451221943, "reward_judge_quality_mean": 0.4312499761581421, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.48836007714271545, "reward_total_composite_std": 0.09807529300451279} {"timestamp_utc": "2026-04-13T07:47:57Z", "mode": "train", "global_step": 193, "epoch": 0.019387242591662482, "loss": -0.1608, "grad_norm": 13.815486907958984, "learning_rate": 9.418181818181818e-06, "num_tokens": 355910.0, "completions/mean_length": 54.5, "completions/min_length": 17.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.5, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.6727335453033447, "rewards/meter/std": 0.29759544134140015, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.2357022762298584, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9905915260314941, "rewards/repeat_soft/std": 0.013315845280885696, "rewards/judge_quality/mean": 0.3812499940395355, "rewards/judge_quality/std": 0.08166787773370743, "rewards/total_composite/mean": 0.4954334795475006, "rewards/total_composite/std": 0.09651092439889908, "reward": 0.4954334795475006, "reward_std": 0.09651093184947968, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24432682991027832, "sampling/sampling_logp_difference/max": 1.2650394439697266, "sampling/importance_sampling_ratio/min": 0.28222814202308655, "sampling/importance_sampling_ratio/mean": 1.0481576919555664, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.2915859818458557, "clip_ratio/low_mean": 0.10347050474956632, "clip_ratio/low_min": 0.10347050474956632, "clip_ratio/high_mean": 0.08240798488259315, "clip_ratio/high_max": 0.08240798488259315, "clip_ratio/region_mean": 0.18587848963215947, "reward_total_mean": 0.4954334795475006, "reward_meter_mean": 0.6727335453033447, "reward_meter_std": 0.29759544134140015, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.2357022762298584, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9905915260314941, "reward_repeat_soft_std": 0.013315845280885696, "reward_judge_quality_mean": 0.3812499940395355, "reward_judge_quality_std": 0.08166787773370743, "reward_total_composite_mean": 0.4954334795475006, "reward_total_composite_std": 0.09651092439889908} {"timestamp_utc": "2026-04-13T07:48:04Z", "mode": "train", "global_step": 194, "epoch": 0.019487694625816173, "loss": 0.2029, "grad_norm": 12.876761436462402, "learning_rate": 9.415151515151515e-06, "num_tokens": 357585.0, "completions/mean_length": 52.375, "completions/min_length": 39.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.375, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.7496733665466309, "rewards/meter/std": 0.36694076657295227, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9982771873474121, "rewards/repeat_soft/std": 0.004632346797734499, "rewards/judge_quality/mean": 0.5137500166893005, "rewards/judge_quality/std": 0.21387162804603577, "rewards/total_composite/mean": 0.5776740908622742, "rewards/total_composite/std": 0.1355331391096115, "reward": 0.5776740908622742, "reward_std": 0.13553312420845032, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20649906992912292, "sampling/sampling_logp_difference/max": 1.2183427810668945, "sampling/importance_sampling_ratio/min": 0.29571983218193054, "sampling/importance_sampling_ratio/mean": 1.0395499467849731, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.77356718480587, "clip_ratio/low_mean": 0.06917979381978512, "clip_ratio/low_min": 0.06917979381978512, "clip_ratio/high_mean": 0.12279230542480946, "clip_ratio/high_max": 0.12279230542480946, "clip_ratio/region_mean": 0.19197209924459457, "reward_total_mean": 0.5776740908622742, "reward_meter_mean": 0.7496733665466309, "reward_meter_std": 0.36694076657295227, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9982771873474121, "reward_repeat_soft_std": 0.004632346797734499, "reward_judge_quality_mean": 0.5137500166893005, "reward_judge_quality_std": 0.21387162804603577, "reward_total_composite_mean": 0.5776740908622742, "reward_total_composite_std": 0.1355331391096115} {"timestamp_utc": "2026-04-13T07:48:12Z", "mode": "train", "global_step": 195, "epoch": 0.019588146659969864, "loss": -0.0291, "grad_norm": 12.050610542297363, "learning_rate": 9.412121212121212e-06, "num_tokens": 359238.0, "completions/mean_length": 47.625, "completions/min_length": 35.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.625, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.2897016704082489, "rewards/meter/std": 0.3196074366569519, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.998386025428772, "rewards/repeat_soft/std": 0.0037448338698595762, "rewards/judge_quality/mean": 0.48624998331069946, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.34002625942230225, "rewards/total_composite/std": 0.2229786217212677, "reward": 0.34002625942230225, "reward_std": 0.2229786068201065, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23713691532611847, "sampling/sampling_logp_difference/max": 1.4660263061523438, "sampling/importance_sampling_ratio/min": 0.23084096610546112, "sampling/importance_sampling_ratio/mean": 1.0835492610931396, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.733154684305191, "clip_ratio/low_mean": 0.0436507947742939, "clip_ratio/low_min": 0.0436507947742939, "clip_ratio/high_mean": 0.16215167567133904, "clip_ratio/high_max": 0.16215167567133904, "clip_ratio/region_mean": 0.20580247044563293, "reward_total_mean": 0.34002625942230225, "reward_meter_mean": 0.2897016704082489, "reward_meter_std": 0.3196074366569519, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.998386025428772, "reward_repeat_soft_std": 0.0037448338698595762, "reward_judge_quality_mean": 0.48624998331069946, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.34002625942230225, "reward_total_composite_std": 0.2229786217212677} {"timestamp_utc": "2026-04-13T07:48:18Z", "mode": "train", "global_step": 196, "epoch": 0.019688598694123555, "loss": 0.002, "grad_norm": 21.040252685546875, "learning_rate": 9.40909090909091e-06, "num_tokens": 360614.0, "completions/mean_length": 23.0, "completions/min_length": 19.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.0, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.6609889268875122, "rewards/meter/std": 0.37576085329055786, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9615821838378906, "rewards/repeat_soft/std": 0.002596014179289341, "rewards/judge_quality/mean": 0.5275000333786011, "rewards/judge_quality/std": 0.2499571591615677, "rewards/total_composite/mean": 0.5649632215499878, "rewards/total_composite/std": 0.1667466014623642, "reward": 0.5649632215499878, "reward_std": 0.1667466014623642, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18898744881153107, "sampling/sampling_logp_difference/max": 1.400949478149414, "sampling/importance_sampling_ratio/min": 0.24636293947696686, "sampling/importance_sampling_ratio/mean": 1.0581945180892944, "sampling/importance_sampling_ratio/max": 1.849361538887024, "entropy": 1.814273864030838, "clip_ratio/low_mean": 0.06732456292957067, "clip_ratio/low_min": 0.06732456292957067, "clip_ratio/high_mean": 0.06317934766411781, "clip_ratio/high_max": 0.06317934766411781, "clip_ratio/region_mean": 0.1305039105936885, "reward_total_mean": 0.5649632215499878, "reward_meter_mean": 0.6609889268875122, "reward_meter_std": 0.37576085329055786, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9615821838378906, "reward_repeat_soft_std": 0.002596014179289341, "reward_judge_quality_mean": 0.5275000333786011, "reward_judge_quality_std": 0.2499571591615677, "reward_total_composite_mean": 0.5649632215499878, "reward_total_composite_std": 0.1667466014623642} {"timestamp_utc": "2026-04-13T07:48:24Z", "mode": "train", "global_step": 197, "epoch": 0.019789050728277247, "loss": 0.0144, "grad_norm": 14.87818717956543, "learning_rate": 9.406060606060607e-06, "num_tokens": 362018.0, "completions/mean_length": 28.5, "completions/min_length": 22.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.5, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.7591187953948975, "rewards/meter/std": 0.3778837025165558, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9472886323928833, "rewards/repeat_soft/std": 0.033697858452796936, "rewards/judge_quality/mean": 0.71875, "rewards/judge_quality/std": 0.2992102801799774, "rewards/total_composite/mean": 0.7138622999191284, "rewards/total_composite/std": 0.24715933203697205, "reward": 0.7138622999191284, "reward_std": 0.24715931713581085, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17814406752586365, "sampling/sampling_logp_difference/max": 1.0547475814819336, "sampling/importance_sampling_ratio/min": 0.34828031063079834, "sampling/importance_sampling_ratio/mean": 1.0341427326202393, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9291309788823128, "clip_ratio/low_mean": 0.08132582809776068, "clip_ratio/low_min": 0.08132582809776068, "clip_ratio/high_mean": 0.10434755031019449, "clip_ratio/high_max": 0.10434755031019449, "clip_ratio/region_mean": 0.18567337840795517, "reward_total_mean": 0.7138622999191284, "reward_meter_mean": 0.7591187953948975, "reward_meter_std": 0.3778837025165558, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9472886323928833, "reward_repeat_soft_std": 0.033697858452796936, "reward_judge_quality_mean": 0.71875, "reward_judge_quality_std": 0.2992102801799774, "reward_total_composite_mean": 0.7138622999191284, "reward_total_composite_std": 0.24715933203697205} {"timestamp_utc": "2026-04-13T07:48:31Z", "mode": "train", "global_step": 198, "epoch": 0.019889502762430938, "loss": 0.0672, "grad_norm": 15.537421226501465, "learning_rate": 9.403030303030304e-06, "num_tokens": 363963.0, "completions/mean_length": 60.125, "completions/min_length": 51.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.125, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.5443116426467896, "rewards/meter/std": 0.37749719619750977, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9726038575172424, "rewards/repeat_soft/std": 0.05000071972608566, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.5973585247993469, "rewards/total_composite/std": 0.20080992579460144, "reward": 0.5973585247993469, "reward_std": 0.20080992579460144, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18356436491012573, "sampling/sampling_logp_difference/max": 2.705568313598633, "sampling/importance_sampling_ratio/min": 0.06683232635259628, "sampling/importance_sampling_ratio/mean": 1.002237319946289, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0010015740990639, "clip_ratio/low_mean": 0.11279523558914661, "clip_ratio/low_min": 0.11279523558914661, "clip_ratio/high_mean": 0.06206165812909603, "clip_ratio/high_max": 0.06206165812909603, "clip_ratio/region_mean": 0.17485689371824265, "reward_total_mean": 0.5973585247993469, "reward_meter_mean": 0.5443116426467896, "reward_meter_std": 0.37749719619750977, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9726038575172424, "reward_repeat_soft_std": 0.05000071972608566, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.5973585247993469, "reward_total_composite_std": 0.20080992579460144} {"timestamp_utc": "2026-04-13T07:48:38Z", "mode": "train", "global_step": 199, "epoch": 0.019989954796584632, "loss": 0.0029, "grad_norm": 23.622743606567383, "learning_rate": 9.4e-06, "num_tokens": 365314.0, "completions/mean_length": 23.875, "completions/min_length": 18.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.875, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.5365124940872192, "rewards/meter/std": 0.39426350593566895, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.3799999952316284, "rewards/judge_quality/std": 0.11501552164554596, "rewards/total_composite/mean": 0.40696874260902405, "rewards/total_composite/std": 0.1946464478969574, "reward": 0.40696874260902405, "reward_std": 0.1946464478969574, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23545973002910614, "sampling/sampling_logp_difference/max": 1.2556962966918945, "sampling/importance_sampling_ratio/min": 0.2848774194717407, "sampling/importance_sampling_ratio/mean": 1.0553559064865112, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6063058376312256, "clip_ratio/low_mean": 0.16020440869033337, "clip_ratio/low_min": 0.16020440869033337, "clip_ratio/high_mean": 0.09125663712620735, "clip_ratio/high_max": 0.09125663712620735, "clip_ratio/region_mean": 0.2514610458165407, "reward_total_mean": 0.40696874260902405, "reward_meter_mean": 0.5365124940872192, "reward_meter_std": 0.39426350593566895, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.3799999952316284, "reward_judge_quality_std": 0.11501552164554596, "reward_total_composite_mean": 0.40696874260902405, "reward_total_composite_std": 0.1946464478969574} {"timestamp_utc": "2026-04-13T07:48:46Z", "mode": "train", "global_step": 200, "epoch": 0.020090406830738324, "loss": 0.0264, "grad_norm": 6.864881992340088, "learning_rate": 9.396969696969697e-06, "num_tokens": 367566.0, "completions/mean_length": 107.5, "completions/min_length": 93.0, "completions/max_length": 136.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.5, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 136.0, "rewards/meter/mean": 0.8985656499862671, "rewards/meter/std": 0.15440396964550018, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9986597299575806, "rewards/repeat_soft/std": 0.0012854236410930753, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.21357084810733795, "rewards/total_composite/mean": 0.6393650770187378, "rewards/total_composite/std": 0.13178274035453796, "reward": 0.6393650770187378, "reward_std": 0.13178272545337677, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2367682307958603, "sampling/sampling_logp_difference/max": 1.4644136428833008, "sampling/importance_sampling_ratio/min": 0.2312135398387909, "sampling/importance_sampling_ratio/mean": 1.041608214378357, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.4034773111343384, "clip_ratio/low_mean": 0.1424694685265422, "clip_ratio/low_min": 0.1424694685265422, "clip_ratio/high_mean": 0.027499999850988388, "clip_ratio/high_max": 0.027499999850988388, "clip_ratio/region_mean": 0.16996946837753057, "reward_total_mean": 0.6393650770187378, "reward_meter_mean": 0.8985656499862671, "reward_meter_std": 0.15440396964550018, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9986597299575806, "reward_repeat_soft_std": 0.0012854236410930753, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.21357084810733795, "reward_total_composite_mean": 0.6393650770187378, "reward_total_composite_std": 0.13178274035453796} {"timestamp_utc": "2026-04-13T07:49:35Z", "mode": "eval", "global_step": 200, "epoch": 0.020090406830738324, "eval_loss": NaN, "eval_runtime": 49.2826, "eval_samples_per_second": 1.623, "eval_steps_per_second": 0.203, "eval_num_tokens": 367566.0, "eval_completions/mean_length": 73.6125, "eval_completions/min_length": 29.2, "eval_completions/max_length": 134.9, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 73.6125, "eval_completions/min_terminated_length": 29.2, "eval_completions/max_terminated_length": 134.9, "eval_rewards/meter/mean": 0.5391432881355286, "eval_rewards/meter/std": 0.38314679861068723, "eval_rewards/count_adherence/mean": 0.970000010728836, "eval_rewards/count_adherence/std": 0.0788449503481388, "eval_rewards/hard_gate/mean": 0.925, "eval_rewards/hard_gate/std": 0.18771235942840575, "eval_rewards/repeat_soft/mean": 0.9907893419265748, "eval_rewards/repeat_soft/std": 0.012920224107801914, "eval_rewards/judge_quality/mean": 0.4497499972581863, "eval_rewards/judge_quality/std": 0.15556477615609765, "eval_rewards/total_composite/mean": 0.4672420650720596, "eval_rewards/total_composite/std": 0.18860110267996788, "eval_reward": 0.4672420650720596, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.16620644852519034, "eval_sampling/sampling_logp_difference/max": 1.0829068183898927, "eval_sampling/importance_sampling_ratio/min": 0.34725173711776736, "eval_sampling/importance_sampling_ratio/mean": 1.0519151449203492, "eval_sampling/importance_sampling_ratio/max": 1.6473710417747498, "eval_entropy": 2.6686283588409423, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.4672420650720596, "eval_reward_meter_mean": 0.5391432881355286, "eval_reward_meter_std": 0.38314679861068723, "eval_reward_count_adherence_mean": 0.970000010728836, "eval_reward_count_adherence_std": 0.0788449503481388, "eval_reward_hard_gate_mean": 0.925, "eval_reward_hard_gate_std": 0.18771235942840575, "eval_reward_repeat_soft_mean": 0.9907893419265748, "eval_reward_repeat_soft_std": 0.012920224107801914, "eval_reward_judge_quality_mean": 0.4497499972581863, "eval_reward_judge_quality_std": 0.15556477615609765, "eval_reward_total_composite_mean": 0.4672420650720596, "eval_reward_total_composite_std": 0.18860110267996788} {"timestamp_utc": "2026-04-13T07:49:46Z", "mode": "train", "global_step": 201, "epoch": 0.020190858864892015, "loss": -0.1181, "grad_norm": 10.919523239135742, "learning_rate": 9.393939393939396e-06, "num_tokens": 369668.0, "completions/mean_length": 73.75, "completions/min_length": 35.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.75, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.3700682520866394, "rewards/meter/std": 0.3486845791339874, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.997421145439148, "rewards/repeat_soft/std": 0.0027841662522405386, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.4321563243865967, "rewards/total_composite/std": 0.10706186294555664, "reward": 0.4321563243865967, "reward_std": 0.10706186294555664, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21561789512634277, "sampling/sampling_logp_difference/max": 1.2610321044921875, "sampling/importance_sampling_ratio/min": 0.28336140513420105, "sampling/importance_sampling_ratio/mean": 1.0400891304016113, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.015211820602417, "clip_ratio/low_mean": 0.138279652222991, "clip_ratio/low_min": 0.138279652222991, "clip_ratio/high_mean": 0.07148056291043758, "clip_ratio/high_max": 0.07148056291043758, "clip_ratio/region_mean": 0.20976021513342857, "reward_total_mean": 0.4321563243865967, "reward_meter_mean": 0.3700682520866394, "reward_meter_std": 0.3486845791339874, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.997421145439148, "reward_repeat_soft_std": 0.0027841662522405386, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.4321563243865967, "reward_total_composite_std": 0.10706186294555664} {"timestamp_utc": "2026-04-13T07:49:54Z", "mode": "train", "global_step": 202, "epoch": 0.020291310899045706, "loss": 0.0434, "grad_norm": 11.686323165893555, "learning_rate": 9.390909090909092e-06, "num_tokens": 371570.0, "completions/mean_length": 65.75, "completions/min_length": 57.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.75, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.9234566688537598, "rewards/meter/std": 0.1139838844537735, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9957336783409119, "rewards/repeat_soft/std": 0.005028849001973867, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.12906256318092346, "rewards/total_composite/mean": 0.5512813329696655, "rewards/total_composite/std": 0.23752687871456146, "reward": 0.5512813329696655, "reward_std": 0.23752686381340027, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2418219894170761, "sampling/sampling_logp_difference/max": 1.222823143005371, "sampling/importance_sampling_ratio/min": 0.29439786076545715, "sampling/importance_sampling_ratio/mean": 1.0712027549743652, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.4567994475364685, "clip_ratio/low_mean": 0.03870224021375179, "clip_ratio/low_min": 0.03870224021375179, "clip_ratio/high_mean": 0.13332573510706425, "clip_ratio/high_max": 0.13332573510706425, "clip_ratio/region_mean": 0.17202797532081604, "reward_total_mean": 0.5512813329696655, "reward_meter_mean": 0.9234566688537598, "reward_meter_std": 0.1139838844537735, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9957336783409119, "reward_repeat_soft_std": 0.005028849001973867, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.12906256318092346, "reward_total_composite_mean": 0.5512813329696655, "reward_total_composite_std": 0.23752687871456146} {"timestamp_utc": "2026-04-13T07:50:02Z", "mode": "train", "global_step": 203, "epoch": 0.020391762933199397, "loss": -0.0037, "grad_norm": 9.007547378540039, "learning_rate": 9.387878787878789e-06, "num_tokens": 374317.0, "completions/mean_length": 136.375, "completions/min_length": 109.0, "completions/max_length": 198.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 136.375, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 198.0, "rewards/meter/mean": 0.6907858848571777, "rewards/meter/std": 0.3659006953239441, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.999170184135437, "rewards/repeat_soft/std": 0.0015115885762497783, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.4157761335372925, "rewards/total_composite/std": 0.18969878554344177, "reward": 0.4157761335372925, "reward_std": 0.18969878554344177, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22914695739746094, "sampling/sampling_logp_difference/max": 1.5648012161254883, "sampling/importance_sampling_ratio/min": 0.20912957191467285, "sampling/importance_sampling_ratio/mean": 1.0604259967803955, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.437682628631592, "clip_ratio/low_mean": 0.07899650000035763, "clip_ratio/low_min": 0.07899650000035763, "clip_ratio/high_mean": 0.11292179115116596, "clip_ratio/high_max": 0.11292179115116596, "clip_ratio/region_mean": 0.1919182911515236, "reward_total_mean": 0.4157761335372925, "reward_meter_mean": 0.6907858848571777, "reward_meter_std": 0.3659006953239441, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.999170184135437, "reward_repeat_soft_std": 0.0015115885762497783, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.4157761335372925, "reward_total_composite_std": 0.18969878554344177} {"timestamp_utc": "2026-04-13T07:50:10Z", "mode": "train", "global_step": 204, "epoch": 0.020492214967353088, "loss": 0.1175, "grad_norm": 18.759410858154297, "learning_rate": 9.384848484848486e-06, "num_tokens": 376023.0, "completions/mean_length": 42.25, "completions/min_length": 34.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.25, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.7116919755935669, "rewards/meter/std": 0.38788291811943054, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.996666431427002, "rewards/repeat_soft/std": 0.005294432397931814, "rewards/judge_quality/mean": 0.5687500238418579, "rewards/judge_quality/std": 0.2568177580833435, "rewards/total_composite/mean": 0.6003146171569824, "rewards/total_composite/std": 0.18185506761074066, "reward": 0.6003146171569824, "reward_std": 0.18185506761074066, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.25824519991874695, "sampling/sampling_logp_difference/max": 1.8854713439941406, "sampling/importance_sampling_ratio/min": 0.15175750851631165, "sampling/importance_sampling_ratio/mean": 1.0785658359527588, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.1049128472805023, "clip_ratio/low_mean": 0.10476029291749, "clip_ratio/low_min": 0.10476029291749, "clip_ratio/high_mean": 0.11218062974512577, "clip_ratio/high_max": 0.11218062974512577, "clip_ratio/region_mean": 0.21694092266261578, "reward_total_mean": 0.6003146171569824, "reward_meter_mean": 0.7116919755935669, "reward_meter_std": 0.38788291811943054, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.996666431427002, "reward_repeat_soft_std": 0.005294432397931814, "reward_judge_quality_mean": 0.5687500238418579, "reward_judge_quality_std": 0.2568177580833435, "reward_total_composite_mean": 0.6003146171569824, "reward_total_composite_std": 0.18185506761074066} {"timestamp_utc": "2026-04-13T07:50:17Z", "mode": "train", "global_step": 205, "epoch": 0.02059266700150678, "loss": 0.1287, "grad_norm": 9.911831855773926, "learning_rate": 9.381818181818183e-06, "num_tokens": 378179.0, "completions/mean_length": 96.5, "completions/min_length": 76.0, "completions/max_length": 135.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.5, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 135.0, "rewards/meter/mean": 0.8738713264465332, "rewards/meter/std": 0.12859904766082764, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9973481893539429, "rewards/repeat_soft/std": 0.00395160960033536, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.6006252765655518, "rewards/total_composite/std": 0.08549527823925018, "reward": 0.6006252765655518, "reward_std": 0.08549528568983078, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2232775241136551, "sampling/sampling_logp_difference/max": 1.452596664428711, "sampling/importance_sampling_ratio/min": 0.23396198451519012, "sampling/importance_sampling_ratio/mean": 1.062524437904358, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.4127989411354065, "clip_ratio/low_mean": 0.0587850222364068, "clip_ratio/low_min": 0.0587850222364068, "clip_ratio/high_mean": 0.13522934913635254, "clip_ratio/high_max": 0.13522934913635254, "clip_ratio/region_mean": 0.19401437137275934, "reward_total_mean": 0.6006252765655518, "reward_meter_mean": 0.8738713264465332, "reward_meter_std": 0.12859904766082764, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9973481893539429, "reward_repeat_soft_std": 0.00395160960033536, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.6006252765655518, "reward_total_composite_std": 0.08549527823925018} {"timestamp_utc": "2026-04-13T07:50:24Z", "mode": "train", "global_step": 206, "epoch": 0.02069311903566047, "loss": 0.1188, "grad_norm": 14.01938247680664, "learning_rate": 9.378787878787879e-06, "num_tokens": 379858.0, "completions/mean_length": 49.875, "completions/min_length": 35.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.875, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.3101171553134918, "rewards/meter/std": 0.19971781969070435, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9820109605789185, "rewards/repeat_soft/std": 0.016810083761811256, "rewards/judge_quality/mean": 0.5024999976158142, "rewards/judge_quality/std": 0.21224987506866455, "rewards/total_composite/mean": 0.46407532691955566, "rewards/total_composite/std": 0.10568965971469879, "reward": 0.46407532691955566, "reward_std": 0.10568965971469879, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2285691648721695, "sampling/sampling_logp_difference/max": 1.982824683189392, "sampling/importance_sampling_ratio/min": 0.137679785490036, "sampling/importance_sampling_ratio/mean": 1.0584670305252075, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4913979172706604, "clip_ratio/low_mean": 0.10394087061285973, "clip_ratio/low_min": 0.10394087061285973, "clip_ratio/high_mean": 0.08304168842732906, "clip_ratio/high_max": 0.08304168842732906, "clip_ratio/region_mean": 0.1869825590401888, "reward_total_mean": 0.46407532691955566, "reward_meter_mean": 0.3101171553134918, "reward_meter_std": 0.19971781969070435, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9820109605789185, "reward_repeat_soft_std": 0.016810083761811256, "reward_judge_quality_mean": 0.5024999976158142, "reward_judge_quality_std": 0.21224987506866455, "reward_total_composite_mean": 0.46407532691955566, "reward_total_composite_std": 0.10568965971469879} {"timestamp_utc": "2026-04-13T07:50:32Z", "mode": "train", "global_step": 207, "epoch": 0.020793571069814165, "loss": 0.1399, "grad_norm": 16.71153450012207, "learning_rate": 9.375757575757576e-06, "num_tokens": 381676.0, "completions/mean_length": 58.25, "completions/min_length": 50.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.25, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.4424096941947937, "rewards/meter/std": 0.311807781457901, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9986618161201477, "rewards/repeat_soft/std": 0.0023097151424735785, "rewards/judge_quality/mean": 0.5575000047683716, "rewards/judge_quality/std": 0.19955310225486755, "rewards/total_composite/mean": 0.5327742099761963, "rewards/total_composite/std": 0.18730784952640533, "reward": 0.5327742099761963, "reward_std": 0.18730784952640533, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.25211960077285767, "sampling/sampling_logp_difference/max": 2.9538486003875732, "sampling/importance_sampling_ratio/min": 0.05213866010308266, "sampling/importance_sampling_ratio/mean": 0.9958014488220215, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1974932253360748, "clip_ratio/low_mean": 0.13629101030528545, "clip_ratio/low_min": 0.13629101030528545, "clip_ratio/high_mean": 0.04890734329819679, "clip_ratio/high_max": 0.04890734329819679, "clip_ratio/region_mean": 0.18519835360348225, "reward_total_mean": 0.5327742099761963, "reward_meter_mean": 0.4424096941947937, "reward_meter_std": 0.311807781457901, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9986618161201477, "reward_repeat_soft_std": 0.0023097151424735785, "reward_judge_quality_mean": 0.5575000047683716, "reward_judge_quality_std": 0.19955310225486755, "reward_total_composite_mean": 0.5327742099761963, "reward_total_composite_std": 0.18730784952640533} {"timestamp_utc": "2026-04-13T07:50:38Z", "mode": "train", "global_step": 208, "epoch": 0.020894023103967856, "loss": -0.1236, "grad_norm": 19.451038360595703, "learning_rate": 9.372727272727273e-06, "num_tokens": 383076.0, "completions/mean_length": 26.0, "completions/min_length": 18.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.0, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.6434499025344849, "rewards/meter/std": 0.40879666805267334, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.952542781829834, "rewards/repeat_soft/std": 0.011327398009598255, "rewards/judge_quality/mean": 0.4087499976158142, "rewards/judge_quality/std": 0.10507649928331375, "rewards/total_composite/mean": 0.5123113393783569, "rewards/total_composite/std": 0.12489007413387299, "reward": 0.5123113393783569, "reward_std": 0.12489006668329239, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1586938351392746, "sampling/sampling_logp_difference/max": 1.4763026237487793, "sampling/importance_sampling_ratio/min": 0.22848092019557953, "sampling/importance_sampling_ratio/mean": 1.011419653892517, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8310238495469093, "clip_ratio/low_mean": 0.09997650468721986, "clip_ratio/low_min": 0.09997650468721986, "clip_ratio/high_mean": 0.05958451982587576, "clip_ratio/high_max": 0.05958451982587576, "clip_ratio/region_mean": 0.15956102451309562, "reward_total_mean": 0.5123113393783569, "reward_meter_mean": 0.6434499025344849, "reward_meter_std": 0.40879666805267334, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.952542781829834, "reward_repeat_soft_std": 0.011327398009598255, "reward_judge_quality_mean": 0.4087499976158142, "reward_judge_quality_std": 0.10507649928331375, "reward_total_composite_mean": 0.5123113393783569, "reward_total_composite_std": 0.12489007413387299} {"timestamp_utc": "2026-04-13T07:50:45Z", "mode": "train", "global_step": 209, "epoch": 0.020994475138121547, "loss": 0.0732, "grad_norm": 11.658206939697266, "learning_rate": 9.36969696969697e-06, "num_tokens": 384645.0, "completions/mean_length": 46.125, "completions/min_length": 37.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.125, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.31836074590682983, "rewards/meter/std": 0.3821393847465515, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9982406497001648, "rewards/repeat_soft/std": 0.004306571092456579, "rewards/judge_quality/mean": 0.6325000524520874, "rewards/judge_quality/std": 0.2474873960018158, "rewards/total_composite/mean": 0.49626511335372925, "rewards/total_composite/std": 0.19499576091766357, "reward": 0.49626511335372925, "reward_std": 0.19499576091766357, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19146274030208588, "sampling/sampling_logp_difference/max": 1.2691144943237305, "sampling/importance_sampling_ratio/min": 0.281080424785614, "sampling/importance_sampling_ratio/mean": 1.0481432676315308, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.180514022707939, "clip_ratio/low_mean": 0.07780514191836119, "clip_ratio/low_min": 0.07780514191836119, "clip_ratio/high_mean": 0.06334707047790289, "clip_ratio/high_max": 0.06334707047790289, "clip_ratio/region_mean": 0.14115221239626408, "reward_total_mean": 0.49626511335372925, "reward_meter_mean": 0.31836074590682983, "reward_meter_std": 0.3821393847465515, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9982406497001648, "reward_repeat_soft_std": 0.004306571092456579, "reward_judge_quality_mean": 0.6325000524520874, "reward_judge_quality_std": 0.2474873960018158, "reward_total_composite_mean": 0.49626511335372925, "reward_total_composite_std": 0.19499576091766357} {"timestamp_utc": "2026-04-13T07:50:52Z", "mode": "train", "global_step": 210, "epoch": 0.02109492717227524, "loss": 0.1079, "grad_norm": 8.990679740905762, "learning_rate": 9.366666666666668e-06, "num_tokens": 386866.0, "completions/mean_length": 112.625, "completions/min_length": 88.0, "completions/max_length": 131.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.625, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.3633342683315277, "rewards/meter/std": 0.30514103174209595, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9887441396713257, "rewards/repeat_soft/std": 0.02644232101738453, "rewards/judge_quality/mean": 0.41499999165534973, "rewards/judge_quality/std": 0.14520922303199768, "rewards/total_composite/mean": 0.3910030722618103, "rewards/total_composite/std": 0.1865619570016861, "reward": 0.3910030722618103, "reward_std": 0.1865619570016861, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2292454093694687, "sampling/sampling_logp_difference/max": 1.7594289779663086, "sampling/importance_sampling_ratio/min": 0.1721431314945221, "sampling/importance_sampling_ratio/mean": 1.053958535194397, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.2477812469005585, "clip_ratio/low_mean": 0.0655823927372694, "clip_ratio/low_min": 0.0655823927372694, "clip_ratio/high_mean": 0.12628458999097347, "clip_ratio/high_max": 0.12628458999097347, "clip_ratio/region_mean": 0.19186698272824287, "reward_total_mean": 0.3910030722618103, "reward_meter_mean": 0.3633342683315277, "reward_meter_std": 0.30514103174209595, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9887441396713257, "reward_repeat_soft_std": 0.02644232101738453, "reward_judge_quality_mean": 0.41499999165534973, "reward_judge_quality_std": 0.14520922303199768, "reward_total_composite_mean": 0.3910030722618103, "reward_total_composite_std": 0.1865619570016861} {"timestamp_utc": "2026-04-13T07:50:58Z", "mode": "train", "global_step": 211, "epoch": 0.02119537920642893, "loss": 0.0086, "grad_norm": 21.03252601623535, "learning_rate": 9.363636363636365e-06, "num_tokens": 388355.0, "completions/mean_length": 26.125, "completions/min_length": 18.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.125, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.7306721210479736, "rewards/meter/std": 0.43190711736679077, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.42624998092651367, "rewards/judge_quality/std": 0.21980105340480804, "rewards/total_composite/mean": 0.5577149987220764, "rewards/total_composite/std": 0.1788819283246994, "reward": 0.5577149987220764, "reward_std": 0.17888189852237701, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20815829932689667, "sampling/sampling_logp_difference/max": 1.2290782928466797, "sampling/importance_sampling_ratio/min": 0.2925620973110199, "sampling/importance_sampling_ratio/mean": 1.0683070421218872, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.57223704457283, "clip_ratio/low_mean": 0.12673186231404543, "clip_ratio/low_min": 0.12673186231404543, "clip_ratio/high_mean": 0.1146390400826931, "clip_ratio/high_max": 0.1146390400826931, "clip_ratio/region_mean": 0.24137090239673853, "reward_total_mean": 0.5577149987220764, "reward_meter_mean": 0.7306721210479736, "reward_meter_std": 0.43190711736679077, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.42624998092651367, "reward_judge_quality_std": 0.21980105340480804, "reward_total_composite_mean": 0.5577149987220764, "reward_total_composite_std": 0.1788819283246994} {"timestamp_utc": "2026-04-13T07:51:06Z", "mode": "train", "global_step": 212, "epoch": 0.02129583124058262, "loss": 0.2297, "grad_norm": 13.52635669708252, "learning_rate": 9.36060606060606e-06, "num_tokens": 390055.0, "completions/mean_length": 54.5, "completions/min_length": 38.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.5, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.9868708848953247, "rewards/meter/std": 0.004795430693775415, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9918505549430847, "rewards/repeat_soft/std": 0.020045360550284386, "rewards/judge_quality/mean": 0.3474999964237213, "rewards/judge_quality/std": 0.11310549825429916, "rewards/total_composite/mean": 0.5717463493347168, "rewards/total_composite/std": 0.07369498163461685, "reward": 0.5717463493347168, "reward_std": 0.07369498163461685, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24047841131687164, "sampling/sampling_logp_difference/max": 1.0447254180908203, "sampling/importance_sampling_ratio/min": 0.35178840160369873, "sampling/importance_sampling_ratio/mean": 1.0468429327011108, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.4133982360363007, "clip_ratio/low_mean": 0.06530501134693623, "clip_ratio/low_min": 0.06530501134693623, "clip_ratio/high_mean": 0.15588182397186756, "clip_ratio/high_max": 0.15588182397186756, "clip_ratio/region_mean": 0.2211868353188038, "reward_total_mean": 0.5717463493347168, "reward_meter_mean": 0.9868708848953247, "reward_meter_std": 0.004795430693775415, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9918505549430847, "reward_repeat_soft_std": 0.020045360550284386, "reward_judge_quality_mean": 0.3474999964237213, "reward_judge_quality_std": 0.11310549825429916, "reward_total_composite_mean": 0.5717463493347168, "reward_total_composite_std": 0.07369498163461685} {"timestamp_utc": "2026-04-13T07:51:14Z", "mode": "train", "global_step": 213, "epoch": 0.021396283274736312, "loss": 0.0765, "grad_norm": 8.80113697052002, "learning_rate": 9.357575757575758e-06, "num_tokens": 392499.0, "completions/mean_length": 121.5, "completions/min_length": 62.0, "completions/max_length": 157.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 121.5, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 157.0, "rewards/meter/mean": 0.32365942001342773, "rewards/meter/std": 0.33367863297462463, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.1414213478565216, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9969816207885742, "rewards/repeat_soft/std": 0.002814301522448659, "rewards/judge_quality/mean": 0.45249998569488525, "rewards/judge_quality/std": 0.18100909888744354, "rewards/total_composite/mean": 0.45135462284088135, "rewards/total_composite/std": 0.12482835352420807, "reward": 0.45135462284088135, "reward_std": 0.12482836097478867, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21992932260036469, "sampling/sampling_logp_difference/max": 1.7447614669799805, "sampling/importance_sampling_ratio/min": 0.17468665540218353, "sampling/importance_sampling_ratio/mean": 1.0516859292984009, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.712333858013153, "clip_ratio/low_mean": 0.15063858963549137, "clip_ratio/low_min": 0.15063858963549137, "clip_ratio/high_mean": 0.09270507097244263, "clip_ratio/high_max": 0.09270507097244263, "clip_ratio/region_mean": 0.243343660607934, "reward_total_mean": 0.45135462284088135, "reward_meter_mean": 0.32365942001342773, "reward_meter_std": 0.33367863297462463, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.1414213478565216, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9969816207885742, "reward_repeat_soft_std": 0.002814301522448659, "reward_judge_quality_mean": 0.45249998569488525, "reward_judge_quality_std": 0.18100909888744354, "reward_total_composite_mean": 0.45135462284088135, "reward_total_composite_std": 0.12482835352420807} {"timestamp_utc": "2026-04-13T07:51:21Z", "mode": "train", "global_step": 214, "epoch": 0.021496735308890003, "loss": -0.0382, "grad_norm": 15.697773933410645, "learning_rate": 9.354545454545455e-06, "num_tokens": 394082.0, "completions/mean_length": 38.875, "completions/min_length": 33.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.875, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.3175961375236511, "rewards/meter/std": 0.38890060782432556, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.995468020439148, "rewards/repeat_soft/std": 0.0074559105560183525, "rewards/judge_quality/mean": 0.627500057220459, "rewards/judge_quality/std": 0.21952873468399048, "rewards/total_composite/mean": 0.47513413429260254, "rewards/total_composite/std": 0.17617429792881012, "reward": 0.47513413429260254, "reward_std": 0.1761743128299713, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17709468305110931, "sampling/sampling_logp_difference/max": 1.4810428619384766, "sampling/importance_sampling_ratio/min": 0.22740042209625244, "sampling/importance_sampling_ratio/mean": 1.0321656465530396, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6610290557146072, "clip_ratio/low_mean": 0.09056520229205489, "clip_ratio/low_min": 0.09056520229205489, "clip_ratio/high_mean": 0.05239104572683573, "clip_ratio/high_max": 0.05239104572683573, "clip_ratio/region_mean": 0.14295624801889062, "reward_total_mean": 0.47513413429260254, "reward_meter_mean": 0.3175961375236511, "reward_meter_std": 0.38890060782432556, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.995468020439148, "reward_repeat_soft_std": 0.0074559105560183525, "reward_judge_quality_mean": 0.627500057220459, "reward_judge_quality_std": 0.21952873468399048, "reward_total_composite_mean": 0.47513413429260254, "reward_total_composite_std": 0.17617429792881012} {"timestamp_utc": "2026-04-13T07:51:28Z", "mode": "train", "global_step": 215, "epoch": 0.021597187343043698, "loss": 0.1524, "grad_norm": 13.002318382263184, "learning_rate": 9.351515151515152e-06, "num_tokens": 396112.0, "completions/mean_length": 77.75, "completions/min_length": 58.0, "completions/max_length": 104.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.75, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.3979058265686035, "rewards/meter/std": 0.42231327295303345, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9979252219200134, "rewards/repeat_soft/std": 0.001515165320597589, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.39308273792266846, "rewards/total_composite/std": 0.19107317924499512, "reward": 0.39308273792266846, "reward_std": 0.19107317924499512, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.25099608302116394, "sampling/sampling_logp_difference/max": 1.6520500183105469, "sampling/importance_sampling_ratio/min": 0.19165661931037903, "sampling/importance_sampling_ratio/mean": 1.0474159717559814, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.917648583650589, "clip_ratio/low_mean": 0.10196277312934399, "clip_ratio/low_min": 0.10196277312934399, "clip_ratio/high_mean": 0.10599080473184586, "clip_ratio/high_max": 0.10599080473184586, "clip_ratio/region_mean": 0.20795357786118984, "reward_total_mean": 0.39308273792266846, "reward_meter_mean": 0.3979058265686035, "reward_meter_std": 0.42231327295303345, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9979252219200134, "reward_repeat_soft_std": 0.001515165320597589, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.39308273792266846, "reward_total_composite_std": 0.19107317924499512} {"timestamp_utc": "2026-04-13T07:51:36Z", "mode": "train", "global_step": 216, "epoch": 0.02169763937719739, "loss": 0.053, "grad_norm": 13.459993362426758, "learning_rate": 9.34848484848485e-06, "num_tokens": 398528.0, "completions/mean_length": 88.0, "completions/min_length": 69.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 88.0, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.2462480366230011, "rewards/meter/std": 0.1796661615371704, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9924381375312805, "rewards/repeat_soft/std": 0.0044843521900475025, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.4137886166572571, "rewards/total_composite/std": 0.050832126289606094, "reward": 0.4137886166572571, "reward_std": 0.0508321188390255, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20468120276927948, "sampling/sampling_logp_difference/max": 2.7741663455963135, "sampling/importance_sampling_ratio/min": 0.06240147724747658, "sampling/importance_sampling_ratio/mean": 1.0105061531066895, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4371421933174133, "clip_ratio/low_mean": 0.0938820093870163, "clip_ratio/low_min": 0.0938820093870163, "clip_ratio/high_mean": 0.09630271978676319, "clip_ratio/high_max": 0.09630271978676319, "clip_ratio/region_mean": 0.1901847291737795, "reward_total_mean": 0.4137886166572571, "reward_meter_mean": 0.2462480366230011, "reward_meter_std": 0.1796661615371704, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9924381375312805, "reward_repeat_soft_std": 0.0044843521900475025, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.4137886166572571, "reward_total_composite_std": 0.050832126289606094} {"timestamp_utc": "2026-04-13T07:51:44Z", "mode": "train", "global_step": 217, "epoch": 0.02179809141135108, "loss": 0.0276, "grad_norm": 9.43420124053955, "learning_rate": 9.345454545454547e-06, "num_tokens": 401007.0, "completions/mean_length": 110.875, "completions/min_length": 50.0, "completions/max_length": 131.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 110.875, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.8257700204849243, "rewards/meter/std": 0.24290962517261505, "rewards/count_adherence/mean": 0.925000011920929, "rewards/count_adherence/std": 0.2121320217847824, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9925748109817505, "rewards/repeat_soft/std": 0.004607753828167915, "rewards/judge_quality/mean": 0.3137499690055847, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.4005802571773529, "rewards/total_composite/std": 0.2571200430393219, "reward": 0.4005802571773529, "reward_std": 0.2571200430393219, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.25337132811546326, "sampling/sampling_logp_difference/max": 1.4746055603027344, "sampling/importance_sampling_ratio/min": 0.2288689911365509, "sampling/importance_sampling_ratio/mean": 1.0746126174926758, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.7519125640392303, "clip_ratio/low_mean": 0.06241595558822155, "clip_ratio/low_min": 0.06241595558822155, "clip_ratio/high_mean": 0.12228629365563393, "clip_ratio/high_max": 0.12228629365563393, "clip_ratio/region_mean": 0.18470224924385548, "reward_total_mean": 0.4005802571773529, "reward_meter_mean": 0.8257700204849243, "reward_meter_std": 0.24290962517261505, "reward_count_adherence_mean": 0.925000011920929, "reward_count_adherence_std": 0.2121320217847824, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9925748109817505, "reward_repeat_soft_std": 0.004607753828167915, "reward_judge_quality_mean": 0.3137499690055847, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.4005802571773529, "reward_total_composite_std": 0.2571200430393219} {"timestamp_utc": "2026-04-13T07:51:52Z", "mode": "train", "global_step": 218, "epoch": 0.02189854344550477, "loss": 0.0009, "grad_norm": 16.42949676513672, "learning_rate": 9.342424242424243e-06, "num_tokens": 402693.0, "completions/mean_length": 42.75, "completions/min_length": 35.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.75, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.5975445508956909, "rewards/meter/std": 0.41694375872612, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9960700869560242, "rewards/repeat_soft/std": 0.007433554623275995, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.5205863118171692, "rewards/total_composite/std": 0.10966836661100388, "reward": 0.5205863118171692, "reward_std": 0.10966836661100388, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22321748733520508, "sampling/sampling_logp_difference/max": 1.2820940017700195, "sampling/importance_sampling_ratio/min": 0.2774556875228882, "sampling/importance_sampling_ratio/mean": 1.075124740600586, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6578336507081985, "clip_ratio/low_mean": 0.08264311123639345, "clip_ratio/low_min": 0.08264311123639345, "clip_ratio/high_mean": 0.09011243376880884, "clip_ratio/high_max": 0.09011243376880884, "clip_ratio/region_mean": 0.1727555450052023, "reward_total_mean": 0.5205863118171692, "reward_meter_mean": 0.5975445508956909, "reward_meter_std": 0.41694375872612, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9960700869560242, "reward_repeat_soft_std": 0.007433554623275995, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.5205863118171692, "reward_total_composite_std": 0.10966836661100388} {"timestamp_utc": "2026-04-13T07:51:59Z", "mode": "train", "global_step": 219, "epoch": 0.021998995479658463, "loss": 0.0003, "grad_norm": 13.988005638122559, "learning_rate": 9.33939393939394e-06, "num_tokens": 404229.0, "completions/mean_length": 44.0, "completions/min_length": 40.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.0, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.22564613819122314, "rewards/meter/std": 0.2446339875459671, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9994621872901917, "rewards/repeat_soft/std": 0.0015211430145427585, "rewards/judge_quality/mean": 0.65625, "rewards/judge_quality/std": 0.2432481348514557, "rewards/total_composite/mean": 0.46914806962013245, "rewards/total_composite/std": 0.14648151397705078, "reward": 0.46914806962013245, "reward_std": 0.14648151397705078, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18220719695091248, "sampling/sampling_logp_difference/max": 1.2400398254394531, "sampling/importance_sampling_ratio/min": 0.2893727123737335, "sampling/importance_sampling_ratio/mean": 1.0694383382797241, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2795404344797134, "clip_ratio/low_mean": 0.11006227973848581, "clip_ratio/low_min": 0.11006227973848581, "clip_ratio/high_mean": 0.06450820062309504, "clip_ratio/high_max": 0.06450820062309504, "clip_ratio/region_mean": 0.17457048036158085, "reward_total_mean": 0.46914806962013245, "reward_meter_mean": 0.22564613819122314, "reward_meter_std": 0.2446339875459671, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9994621872901917, "reward_repeat_soft_std": 0.0015211430145427585, "reward_judge_quality_mean": 0.65625, "reward_judge_quality_std": 0.2432481348514557, "reward_total_composite_mean": 0.46914806962013245, "reward_total_composite_std": 0.14648151397705078} {"timestamp_utc": "2026-04-13T07:52:06Z", "mode": "train", "global_step": 220, "epoch": 0.022099447513812154, "loss": 0.0078, "grad_norm": 13.550875663757324, "learning_rate": 9.336363636363637e-06, "num_tokens": 405869.0, "completions/mean_length": 40.0, "completions/min_length": 34.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.0, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.46394291520118713, "rewards/meter/std": 0.2540203332901001, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9941651821136475, "rewards/repeat_soft/std": 0.007878880016505718, "rewards/judge_quality/mean": 0.5349999666213989, "rewards/judge_quality/std": 0.24663449823856354, "rewards/total_composite/mean": 0.468347430229187, "rewards/total_composite/std": 0.2445186972618103, "reward": 0.468347430229187, "reward_std": 0.2445186972618103, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20536713302135468, "sampling/sampling_logp_difference/max": 1.563028335571289, "sampling/importance_sampling_ratio/min": 0.20950067043304443, "sampling/importance_sampling_ratio/mean": 1.0111911296844482, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3682081010192633, "clip_ratio/low_mean": 0.12227142415940762, "clip_ratio/low_min": 0.12227142415940762, "clip_ratio/high_mean": 0.06669955886900425, "clip_ratio/high_max": 0.06669955886900425, "clip_ratio/region_mean": 0.18897098302841187, "reward_total_mean": 0.468347430229187, "reward_meter_mean": 0.46394291520118713, "reward_meter_std": 0.2540203332901001, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9941651821136475, "reward_repeat_soft_std": 0.007878880016505718, "reward_judge_quality_mean": 0.5349999666213989, "reward_judge_quality_std": 0.24663449823856354, "reward_total_composite_mean": 0.468347430229187, "reward_total_composite_std": 0.2445186972618103} {"timestamp_utc": "2026-04-13T07:52:12Z", "mode": "train", "global_step": 221, "epoch": 0.022199899547965845, "loss": -0.0787, "grad_norm": 22.784671783447266, "learning_rate": 9.333333333333334e-06, "num_tokens": 407234.0, "completions/mean_length": 20.625, "completions/min_length": 15.0, "completions/max_length": 27.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 20.625, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.7635542154312134, "rewards/meter/std": 0.41205763816833496, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8938847780227661, "rewards/repeat_soft/std": 0.16115181148052216, "rewards/judge_quality/mean": 0.4012500047683716, "rewards/judge_quality/std": 0.10260012745857239, "rewards/total_composite/mean": 0.5158623456954956, "rewards/total_composite/std": 0.22194378077983856, "reward": 0.5158623456954956, "reward_std": 0.22194376587867737, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1971156895160675, "sampling/sampling_logp_difference/max": 1.424391746520996, "sampling/importance_sampling_ratio/min": 0.2406548112630844, "sampling/importance_sampling_ratio/mean": 1.0366919040679932, "sampling/importance_sampling_ratio/max": 1.914661169052124, "entropy": 1.931114301085472, "clip_ratio/low_mean": 0.04880952462553978, "clip_ratio/low_min": 0.04880952462553978, "clip_ratio/high_mean": 0.14553836593404412, "clip_ratio/high_max": 0.14553836593404412, "clip_ratio/region_mean": 0.1943478905595839, "reward_total_mean": 0.5158623456954956, "reward_meter_mean": 0.7635542154312134, "reward_meter_std": 0.41205763816833496, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8938847780227661, "reward_repeat_soft_std": 0.16115181148052216, "reward_judge_quality_mean": 0.4012500047683716, "reward_judge_quality_std": 0.10260012745857239, "reward_total_composite_mean": 0.5158623456954956, "reward_total_composite_std": 0.22194378077983856} {"timestamp_utc": "2026-04-13T07:52:19Z", "mode": "train", "global_step": 222, "epoch": 0.02230035158211954, "loss": 0.0696, "grad_norm": 12.268951416015625, "learning_rate": 9.33030303030303e-06, "num_tokens": 408868.0, "completions/mean_length": 41.25, "completions/min_length": 37.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.25, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.24934370815753937, "rewards/meter/std": 0.352595716714859, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9918965101242065, "rewards/repeat_soft/std": 0.008309373632073402, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.41893401741981506, "rewards/total_composite/std": 0.09815803915262222, "reward": 0.41893401741981506, "reward_std": 0.09815803915262222, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19416873157024384, "sampling/sampling_logp_difference/max": 1.2836970090866089, "sampling/importance_sampling_ratio/min": 0.33600616455078125, "sampling/importance_sampling_ratio/mean": 1.062048077583313, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3454811573028564, "clip_ratio/low_mean": 0.13969633169472218, "clip_ratio/low_min": 0.13969633169472218, "clip_ratio/high_mean": 0.04002079088240862, "clip_ratio/high_max": 0.04002079088240862, "clip_ratio/region_mean": 0.1797171225771308, "reward_total_mean": 0.41893401741981506, "reward_meter_mean": 0.24934370815753937, "reward_meter_std": 0.352595716714859, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9918965101242065, "reward_repeat_soft_std": 0.008309373632073402, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.41893401741981506, "reward_total_composite_std": 0.09815803915262222} {"timestamp_utc": "2026-04-13T07:52:27Z", "mode": "train", "global_step": 223, "epoch": 0.02240080361627323, "loss": -0.0104, "grad_norm": 8.55893611907959, "learning_rate": 9.327272727272729e-06, "num_tokens": 411085.0, "completions/mean_length": 105.125, "completions/min_length": 88.0, "completions/max_length": 121.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 105.125, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.7916437983512878, "rewards/meter/std": 0.24882195889949799, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9991984367370605, "rewards/repeat_soft/std": 0.001545512699522078, "rewards/judge_quality/mean": 0.23749999701976776, "rewards/judge_quality/std": 0.09910311549901962, "rewards/total_composite/mean": 0.40125560760498047, "rewards/total_composite/std": 0.16618593037128448, "reward": 0.40125560760498047, "reward_std": 0.16618593037128448, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2330099493265152, "sampling/sampling_logp_difference/max": 1.4484367370605469, "sampling/importance_sampling_ratio/min": 0.23493726551532745, "sampling/importance_sampling_ratio/mean": 1.07094407081604, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.7301617562770844, "clip_ratio/low_mean": 0.025815216824412346, "clip_ratio/low_min": 0.025815216824412346, "clip_ratio/high_mean": 0.16274195350706577, "clip_ratio/high_max": 0.16274195350706577, "clip_ratio/region_mean": 0.18855717033147812, "reward_total_mean": 0.40125560760498047, "reward_meter_mean": 0.7916437983512878, "reward_meter_std": 0.24882195889949799, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9991984367370605, "reward_repeat_soft_std": 0.001545512699522078, "reward_judge_quality_mean": 0.23749999701976776, "reward_judge_quality_std": 0.09910311549901962, "reward_total_composite_mean": 0.40125560760498047, "reward_total_composite_std": 0.16618593037128448} {"timestamp_utc": "2026-04-13T07:52:35Z", "mode": "train", "global_step": 224, "epoch": 0.022501255650426922, "loss": 0.1158, "grad_norm": 13.292996406555176, "learning_rate": 9.324242424242424e-06, "num_tokens": 412778.0, "completions/mean_length": 58.625, "completions/min_length": 48.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.625, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.5885939598083496, "rewards/meter/std": 0.3498261570930481, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.996599018573761, "rewards/repeat_soft/std": 0.0025883002672344446, "rewards/judge_quality/mean": 0.5737500190734863, "rewards/judge_quality/std": 0.1566559225320816, "rewards/total_composite/mean": 0.5756773948669434, "rewards/total_composite/std": 0.1631508320569992, "reward": 0.5756773948669434, "reward_std": 0.1631508320569992, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2243303507566452, "sampling/sampling_logp_difference/max": 1.7579689025878906, "sampling/importance_sampling_ratio/min": 0.17239466309547424, "sampling/importance_sampling_ratio/mean": 1.0411030054092407, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.9070996046066284, "clip_ratio/low_mean": 0.09940476529300213, "clip_ratio/low_min": 0.09940476529300213, "clip_ratio/high_mean": 0.1176231075078249, "clip_ratio/high_max": 0.1176231075078249, "clip_ratio/region_mean": 0.21702787280082703, "reward_total_mean": 0.5756773948669434, "reward_meter_mean": 0.5885939598083496, "reward_meter_std": 0.3498261570930481, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.996599018573761, "reward_repeat_soft_std": 0.0025883002672344446, "reward_judge_quality_mean": 0.5737500190734863, "reward_judge_quality_std": 0.1566559225320816, "reward_total_composite_mean": 0.5756773948669434, "reward_total_composite_std": 0.1631508320569992} {"timestamp_utc": "2026-04-13T07:52:42Z", "mode": "train", "global_step": 225, "epoch": 0.022601707684580613, "loss": 0.0863, "grad_norm": 9.034035682678223, "learning_rate": 9.321212121212122e-06, "num_tokens": 415296.0, "completions/mean_length": 114.75, "completions/min_length": 93.0, "completions/max_length": 146.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 114.75, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 146.0, "rewards/meter/mean": 0.132351815700531, "rewards/meter/std": 0.15599699318408966, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9927990436553955, "rewards/repeat_soft/std": 0.009575942531228065, "rewards/judge_quality/mean": 0.26749998331069946, "rewards/judge_quality/std": 0.10375107079744339, "rewards/total_composite/mean": 0.371489018201828, "rewards/total_composite/std": 0.026845812797546387, "reward": 0.371489018201828, "reward_std": 0.026845814660191536, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24750882387161255, "sampling/sampling_logp_difference/max": 1.3614006042480469, "sampling/importance_sampling_ratio/min": 0.25630155205726624, "sampling/importance_sampling_ratio/mean": 1.0550618171691895, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.477659285068512, "clip_ratio/low_mean": 0.12866520509123802, "clip_ratio/low_min": 0.12866520509123802, "clip_ratio/high_mean": 0.04819867201149464, "clip_ratio/high_max": 0.04819867201149464, "clip_ratio/region_mean": 0.17686387710273266, "reward_total_mean": 0.371489018201828, "reward_meter_mean": 0.132351815700531, "reward_meter_std": 0.15599699318408966, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9927990436553955, "reward_repeat_soft_std": 0.009575942531228065, "reward_judge_quality_mean": 0.26749998331069946, "reward_judge_quality_std": 0.10375107079744339, "reward_total_composite_mean": 0.371489018201828, "reward_total_composite_std": 0.026845812797546387} {"timestamp_utc": "2026-04-13T07:52:49Z", "mode": "train", "global_step": 226, "epoch": 0.022702159718734304, "loss": -0.085, "grad_norm": 15.010398864746094, "learning_rate": 9.318181818181819e-06, "num_tokens": 416665.0, "completions/mean_length": 21.125, "completions/min_length": 16.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.125, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.4514320194721222, "rewards/meter/std": 0.3760293424129486, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.41749998927116394, "rewards/judge_quality/std": 0.06902380287647247, "rewards/total_composite/mean": 0.4698364734649658, "rewards/total_composite/std": 0.11148180067539215, "reward": 0.4698364734649658, "reward_std": 0.11148180812597275, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14198869466781616, "sampling/sampling_logp_difference/max": 0.9297459125518799, "sampling/importance_sampling_ratio/min": 0.39465397596359253, "sampling/importance_sampling_ratio/mean": 1.024768590927124, "sampling/importance_sampling_ratio/max": 1.9370837211608887, "entropy": 0.9989406615495682, "clip_ratio/low_mean": 0.06531892158091068, "clip_ratio/low_min": 0.06531892158091068, "clip_ratio/high_mean": 0.06238553300499916, "clip_ratio/high_max": 0.06238553300499916, "clip_ratio/region_mean": 0.12770445458590984, "reward_total_mean": 0.4698364734649658, "reward_meter_mean": 0.4514320194721222, "reward_meter_std": 0.3760293424129486, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.41749998927116394, "reward_judge_quality_std": 0.06902380287647247, "reward_total_composite_mean": 0.4698364734649658, "reward_total_composite_std": 0.11148180067539215} {"timestamp_utc": "2026-04-13T07:52:57Z", "mode": "train", "global_step": 227, "epoch": 0.022802611752887995, "loss": 0.1018, "grad_norm": 9.615615844726562, "learning_rate": 9.315151515151516e-06, "num_tokens": 418843.0, "completions/mean_length": 96.25, "completions/min_length": 74.0, "completions/max_length": 126.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.25, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 126.0, "rewards/meter/mean": 0.4061456024646759, "rewards/meter/std": 0.3346509337425232, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9986236095428467, "rewards/repeat_soft/std": 0.0010029288241639733, "rewards/judge_quality/mean": 0.2849999964237213, "rewards/judge_quality/std": 0.12177261710166931, "rewards/total_composite/mean": 0.30179980397224426, "rewards/total_composite/std": 0.20693248510360718, "reward": 0.30179980397224426, "reward_std": 0.20693247020244598, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24260981380939484, "sampling/sampling_logp_difference/max": 1.4735374450683594, "sampling/importance_sampling_ratio/min": 0.22911356389522552, "sampling/importance_sampling_ratio/mean": 1.0716917514801025, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 4.097844690084457, "clip_ratio/low_mean": 0.044188953936100006, "clip_ratio/low_min": 0.044188953936100006, "clip_ratio/high_mean": 0.14237085357308388, "clip_ratio/high_max": 0.14237085357308388, "clip_ratio/region_mean": 0.18655980750918388, "reward_total_mean": 0.30179980397224426, "reward_meter_mean": 0.4061456024646759, "reward_meter_std": 0.3346509337425232, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9986236095428467, "reward_repeat_soft_std": 0.0010029288241639733, "reward_judge_quality_mean": 0.2849999964237213, "reward_judge_quality_std": 0.12177261710166931, "reward_total_composite_mean": 0.30179980397224426, "reward_total_composite_std": 0.20693248510360718} {"timestamp_utc": "2026-04-13T07:53:05Z", "mode": "train", "global_step": 228, "epoch": 0.022903063787041687, "loss": 0.0147, "grad_norm": 11.655343055725098, "learning_rate": 9.312121212121212e-06, "num_tokens": 420670.0, "completions/mean_length": 58.375, "completions/min_length": 46.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.375, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.5829670429229736, "rewards/meter/std": 0.42181527614593506, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9993855953216553, "rewards/repeat_soft/std": 0.0017378831980749965, "rewards/judge_quality/mean": 0.3499999940395355, "rewards/judge_quality/std": 0.10690449178218842, "rewards/total_composite/mean": 0.44242537021636963, "rewards/total_composite/std": 0.21695834398269653, "reward": 0.44242537021636963, "reward_std": 0.21695834398269653, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2133873850107193, "sampling/sampling_logp_difference/max": 1.4677143096923828, "sampling/importance_sampling_ratio/min": 0.23045162856578827, "sampling/importance_sampling_ratio/mean": 1.0543785095214844, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.69685398042202, "clip_ratio/low_mean": 0.06226163078099489, "clip_ratio/low_min": 0.06226163078099489, "clip_ratio/high_mean": 0.13080392964184284, "clip_ratio/high_max": 0.13080392964184284, "clip_ratio/region_mean": 0.19306556042283773, "reward_total_mean": 0.44242537021636963, "reward_meter_mean": 0.5829670429229736, "reward_meter_std": 0.42181527614593506, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9993855953216553, "reward_repeat_soft_std": 0.0017378831980749965, "reward_judge_quality_mean": 0.3499999940395355, "reward_judge_quality_std": 0.10690449178218842, "reward_total_composite_mean": 0.44242537021636963, "reward_total_composite_std": 0.21695834398269653} {"timestamp_utc": "2026-04-13T07:53:11Z", "mode": "train", "global_step": 229, "epoch": 0.023003515821195378, "loss": 0.1245, "grad_norm": 17.736083984375, "learning_rate": 9.30909090909091e-06, "num_tokens": 422246.0, "completions/mean_length": 41.0, "completions/min_length": 37.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.0, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9324971437454224, "rewards/meter/std": 0.10176718235015869, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9925810694694519, "rewards/repeat_soft/std": 0.007585481274873018, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.6034588813781738, "rewards/total_composite/std": 0.0284835547208786, "reward": 0.6034588813781738, "reward_std": 0.028483537957072258, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22982312738895416, "sampling/sampling_logp_difference/max": 1.3124065399169922, "sampling/importance_sampling_ratio/min": 0.26917150616645813, "sampling/importance_sampling_ratio/mean": 1.020029067993164, "sampling/importance_sampling_ratio/max": 1.9736405611038208, "entropy": 2.5963164418935776, "clip_ratio/low_mean": 0.04195804335176945, "clip_ratio/low_min": 0.04195804335176945, "clip_ratio/high_mean": 0.18593565560877323, "clip_ratio/high_max": 0.18593565560877323, "clip_ratio/region_mean": 0.22789369896054268, "reward_total_mean": 0.6034588813781738, "reward_meter_mean": 0.9324971437454224, "reward_meter_std": 0.10176718235015869, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9925810694694519, "reward_repeat_soft_std": 0.007585481274873018, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.6034588813781738, "reward_total_composite_std": 0.0284835547208786} {"timestamp_utc": "2026-04-13T07:53:18Z", "mode": "train", "global_step": 230, "epoch": 0.023103967855349072, "loss": 0.0795, "grad_norm": 20.825172424316406, "learning_rate": 9.306060606060608e-06, "num_tokens": 424107.0, "completions/mean_length": 48.625, "completions/min_length": 44.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.625, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.4202841818332672, "rewards/meter/std": 0.26213112473487854, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9920446872711182, "rewards/repeat_soft/std": 0.006530150305479765, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.41952887177467346, "rewards/total_composite/std": 0.17820438742637634, "reward": 0.41952887177467346, "reward_std": 0.17820440232753754, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23581646382808685, "sampling/sampling_logp_difference/max": 1.6607871055603027, "sampling/importance_sampling_ratio/min": 0.1899893879890442, "sampling/importance_sampling_ratio/mean": 1.0112229585647583, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.193833500146866, "clip_ratio/low_mean": 0.01886792480945587, "clip_ratio/low_min": 0.01886792480945587, "clip_ratio/high_mean": 0.22404972184449434, "clip_ratio/high_max": 0.22404972184449434, "clip_ratio/region_mean": 0.24291764665395021, "reward_total_mean": 0.41952887177467346, "reward_meter_mean": 0.4202841818332672, "reward_meter_std": 0.26213112473487854, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9920446872711182, "reward_repeat_soft_std": 0.006530150305479765, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.41952887177467346, "reward_total_composite_std": 0.17820438742637634} {"timestamp_utc": "2026-04-13T07:53:24Z", "mode": "train", "global_step": 231, "epoch": 0.023204419889502764, "loss": 0.1162, "grad_norm": 9.56534194946289, "learning_rate": 9.303030303030303e-06, "num_tokens": 425901.0, "completions/mean_length": 60.25, "completions/min_length": 42.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.25, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9549270868301392, "rewards/meter/std": 0.06464817374944687, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9862501621246338, "rewards/repeat_soft/std": 0.023244677111506462, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.20860078930854797, "rewards/total_composite/mean": 0.6222413182258606, "rewards/total_composite/std": 0.13817496597766876, "reward": 0.6222413182258606, "reward_std": 0.13817498087882996, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21060773730278015, "sampling/sampling_logp_difference/max": 1.1233088970184326, "sampling/importance_sampling_ratio/min": 0.3354046046733856, "sampling/importance_sampling_ratio/mean": 1.0613038539886475, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.8028874695301056, "clip_ratio/low_mean": 0.14069972466677427, "clip_ratio/low_min": 0.14069972466677427, "clip_ratio/high_mean": 0.032738097012043, "clip_ratio/high_max": 0.032738097012043, "clip_ratio/region_mean": 0.17343782167881727, "reward_total_mean": 0.6222413182258606, "reward_meter_mean": 0.9549270868301392, "reward_meter_std": 0.06464817374944687, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9862501621246338, "reward_repeat_soft_std": 0.023244677111506462, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.20860078930854797, "reward_total_composite_mean": 0.6222413182258606, "reward_total_composite_std": 0.13817496597766876} {"timestamp_utc": "2026-04-13T07:53:32Z", "mode": "train", "global_step": 232, "epoch": 0.023304871923656455, "loss": 0.0053, "grad_norm": 10.48536491394043, "learning_rate": 9.3e-06, "num_tokens": 427731.0, "completions/mean_length": 62.75, "completions/min_length": 53.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.75, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.7720727920532227, "rewards/meter/std": 0.38212233781814575, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9976580142974854, "rewards/repeat_soft/std": 0.006362478248775005, "rewards/judge_quality/mean": 0.7325000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.7141427397727966, "rewards/total_composite/std": 0.2544417083263397, "reward": 0.7141427397727966, "reward_std": 0.2544417381286621, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19832520186901093, "sampling/sampling_logp_difference/max": 1.1894540786743164, "sampling/importance_sampling_ratio/min": 0.3043873906135559, "sampling/importance_sampling_ratio/mean": 1.0476423501968384, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.552804797887802, "clip_ratio/low_mean": 0.08357260189950466, "clip_ratio/low_min": 0.08357260189950466, "clip_ratio/high_mean": 0.09559883549809456, "clip_ratio/high_max": 0.09559883549809456, "clip_ratio/region_mean": 0.17917143739759922, "reward_total_mean": 0.7141427397727966, "reward_meter_mean": 0.7720727920532227, "reward_meter_std": 0.38212233781814575, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9976580142974854, "reward_repeat_soft_std": 0.006362478248775005, "reward_judge_quality_mean": 0.7325000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.7141427397727966, "reward_total_composite_std": 0.2544417083263397} {"timestamp_utc": "2026-04-13T07:53:41Z", "mode": "train", "global_step": 233, "epoch": 0.023405323957810146, "loss": 0.0208, "grad_norm": 7.633944511413574, "learning_rate": 9.296969696969698e-06, "num_tokens": 430685.0, "completions/mean_length": 177.25, "completions/min_length": 134.0, "completions/max_length": 232.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 177.25, "completions/min_terminated_length": 134.0, "completions/max_terminated_length": 232.0, "rewards/meter/mean": 0.1322435736656189, "rewards/meter/std": 0.123989999294281, "rewards/count_adherence/mean": 0.9166666269302368, "rewards/count_adherence/std": 0.0890870913863182, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.998924970626831, "rewards/repeat_soft/std": 0.0005966781172901392, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.18431341648101807, "rewards/total_composite/mean": 0.32089686393737793, "rewards/total_composite/std": 0.13752084970474243, "reward": 0.32089686393737793, "reward_std": 0.13752086460590363, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23580877482891083, "sampling/sampling_logp_difference/max": 1.2670583724975586, "sampling/importance_sampling_ratio/min": 0.28165894746780396, "sampling/importance_sampling_ratio/mean": 1.0716066360473633, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.9186656773090363, "clip_ratio/low_mean": 0.03377334773540497, "clip_ratio/low_min": 0.03377334773540497, "clip_ratio/high_mean": 0.15041503123939037, "clip_ratio/high_max": 0.15041503123939037, "clip_ratio/region_mean": 0.18418837897479534, "reward_total_mean": 0.32089686393737793, "reward_meter_mean": 0.1322435736656189, "reward_meter_std": 0.123989999294281, "reward_count_adherence_mean": 0.9166666269302368, "reward_count_adherence_std": 0.0890870913863182, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.998924970626831, "reward_repeat_soft_std": 0.0005966781172901392, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.18431341648101807, "reward_total_composite_mean": 0.32089686393737793, "reward_total_composite_std": 0.13752084970474243} {"timestamp_utc": "2026-04-13T07:53:48Z", "mode": "train", "global_step": 234, "epoch": 0.023505775991963837, "loss": -0.0942, "grad_norm": 17.617843627929688, "learning_rate": 9.293939393939395e-06, "num_tokens": 432140.0, "completions/mean_length": 25.875, "completions/min_length": 18.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.875, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.315799742937088, "rewards/meter/std": 0.3497828245162964, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9580180644989014, "rewards/repeat_soft/std": 0.009412449784576893, "rewards/judge_quality/mean": 0.5437500476837158, "rewards/judge_quality/std": 0.3325631320476532, "rewards/total_composite/mean": 0.48225101828575134, "rewards/total_composite/std": 0.19423244893550873, "reward": 0.48225101828575134, "reward_std": 0.19423244893550873, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20272737741470337, "sampling/sampling_logp_difference/max": 1.5750455856323242, "sampling/importance_sampling_ratio/min": 0.20699812471866608, "sampling/importance_sampling_ratio/mean": 1.0383530855178833, "sampling/importance_sampling_ratio/max": 1.8340553045272827, "entropy": 1.7450096309185028, "clip_ratio/low_mean": 0.07225429499521852, "clip_ratio/low_min": 0.07225429499521852, "clip_ratio/high_mean": 0.09486280009150505, "clip_ratio/high_max": 0.09486280009150505, "clip_ratio/region_mean": 0.16711709508672357, "reward_total_mean": 0.48225101828575134, "reward_meter_mean": 0.315799742937088, "reward_meter_std": 0.3497828245162964, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9580180644989014, "reward_repeat_soft_std": 0.009412449784576893, "reward_judge_quality_mean": 0.5437500476837158, "reward_judge_quality_std": 0.3325631320476532, "reward_total_composite_mean": 0.48225101828575134, "reward_total_composite_std": 0.19423244893550873} {"timestamp_utc": "2026-04-13T07:53:57Z", "mode": "train", "global_step": 235, "epoch": 0.023606228026117528, "loss": 0.0866, "grad_norm": 7.5676469802856445, "learning_rate": 9.29090909090909e-06, "num_tokens": 434776.0, "completions/mean_length": 134.5, "completions/min_length": 82.0, "completions/max_length": 332.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 134.5, "completions/min_terminated_length": 82.0, "completions/max_terminated_length": 332.0, "rewards/meter/mean": 0.5199936628341675, "rewards/meter/std": 0.3432442545890808, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9915440678596497, "rewards/repeat_soft/std": 0.013796639628708363, "rewards/judge_quality/mean": 0.26874998211860657, "rewards/judge_quality/std": 0.13505950570106506, "rewards/total_composite/mean": 0.33347609639167786, "rewards/total_composite/std": 0.22276601195335388, "reward": 0.33347609639167786, "reward_std": 0.22276602685451508, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24162688851356506, "sampling/sampling_logp_difference/max": 1.4961953163146973, "sampling/importance_sampling_ratio/min": 0.22398072481155396, "sampling/importance_sampling_ratio/mean": 1.0915111303329468, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 4.0247719287872314, "clip_ratio/low_mean": 0.043165650218725204, "clip_ratio/low_min": 0.043165650218725204, "clip_ratio/high_mean": 0.14359846152365208, "clip_ratio/high_max": 0.14359846152365208, "clip_ratio/region_mean": 0.18676411174237728, "reward_total_mean": 0.33347609639167786, "reward_meter_mean": 0.5199936628341675, "reward_meter_std": 0.3432442545890808, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9915440678596497, "reward_repeat_soft_std": 0.013796639628708363, "reward_judge_quality_mean": 0.26874998211860657, "reward_judge_quality_std": 0.13505950570106506, "reward_total_composite_mean": 0.33347609639167786, "reward_total_composite_std": 0.22276601195335388} {"timestamp_utc": "2026-04-13T07:54:06Z", "mode": "train", "global_step": 236, "epoch": 0.02370668006027122, "loss": 0.0838, "grad_norm": 11.439743041992188, "learning_rate": 9.28787878787879e-06, "num_tokens": 437304.0, "completions/mean_length": 122.0, "completions/min_length": 104.0, "completions/max_length": 140.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.0, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 140.0, "rewards/meter/mean": 0.6040682792663574, "rewards/meter/std": 0.22797472774982452, "rewards/count_adherence/mean": 0.9166666269302368, "rewards/count_adherence/std": 0.0890870913863182, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9847686886787415, "rewards/repeat_soft/std": 0.017375530675053596, "rewards/judge_quality/mean": 0.5325000286102295, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.5345677137374878, "rewards/total_composite/std": 0.10127469152212143, "reward": 0.5345677137374878, "reward_std": 0.10127468407154083, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23363074660301208, "sampling/sampling_logp_difference/max": 1.8927171230316162, "sampling/importance_sampling_ratio/min": 0.1506618857383728, "sampling/importance_sampling_ratio/mean": 1.0151104927062988, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7888659238815308, "clip_ratio/low_mean": 0.11487327516078949, "clip_ratio/low_min": 0.11487327516078949, "clip_ratio/high_mean": 0.12245976366102695, "clip_ratio/high_max": 0.12245976366102695, "clip_ratio/region_mean": 0.23733303882181644, "reward_total_mean": 0.5345677137374878, "reward_meter_mean": 0.6040682792663574, "reward_meter_std": 0.22797472774982452, "reward_count_adherence_mean": 0.9166666269302368, "reward_count_adherence_std": 0.0890870913863182, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9847686886787415, "reward_repeat_soft_std": 0.017375530675053596, "reward_judge_quality_mean": 0.5325000286102295, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.5345677137374878, "reward_total_composite_std": 0.10127469152212143} {"timestamp_utc": "2026-04-13T07:54:18Z", "mode": "train", "global_step": 237, "epoch": 0.02380713209442491, "loss": -0.1479, "grad_norm": 3.540306568145752, "learning_rate": 9.284848484848485e-06, "num_tokens": 439032.0, "completions/mean_length": 126.0, "completions/min_length": 51.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 70.85714721679688, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.33497804403305054, "rewards/meter/std": 0.40851643681526184, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9910533428192139, "rewards/repeat_soft/std": 0.014115079306066036, "rewards/judge_quality/mean": 0.3100000023841858, "rewards/judge_quality/std": 0.1345893144607544, "rewards/total_composite/mean": 0.38116806745529175, "rewards/total_composite/std": 0.17786787450313568, "reward": 0.38116806745529175, "reward_std": 0.17786787450313568, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23217308521270752, "sampling/sampling_logp_difference/max": 1.4035301208496094, "sampling/importance_sampling_ratio/min": 0.24572797119617462, "sampling/importance_sampling_ratio/mean": 1.089635968208313, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.430901527404785, "clip_ratio/low_mean": 0.055441899225115776, "clip_ratio/low_min": 0.055441899225115776, "clip_ratio/high_mean": 0.11968373786658049, "clip_ratio/high_max": 0.11968373786658049, "clip_ratio/region_mean": 0.17512563709169626, "reward_total_mean": 0.38116806745529175, "reward_meter_mean": 0.33497804403305054, "reward_meter_std": 0.40851643681526184, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9910533428192139, "reward_repeat_soft_std": 0.014115079306066036, "reward_judge_quality_mean": 0.3100000023841858, "reward_judge_quality_std": 0.1345893144607544, "reward_total_composite_mean": 0.38116806745529175, "reward_total_composite_std": 0.17786787450313568} {"timestamp_utc": "2026-04-13T07:54:26Z", "mode": "train", "global_step": 238, "epoch": 0.023907584128578605, "loss": 0.0075, "grad_norm": 10.821329116821289, "learning_rate": 9.281818181818183e-06, "num_tokens": 440710.0, "completions/mean_length": 52.75, "completions/min_length": 45.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.75, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9594722390174866, "rewards/meter/std": 0.06740762293338776, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9954468607902527, "rewards/repeat_soft/std": 0.00810985453426838, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.1440672129392624, "rewards/total_composite/mean": 0.6222023367881775, "rewards/total_composite/std": 0.10977651923894882, "reward": 0.6222023367881775, "reward_std": 0.10977650433778763, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21899792551994324, "sampling/sampling_logp_difference/max": 1.0820188522338867, "sampling/importance_sampling_ratio/min": 0.33891063928604126, "sampling/importance_sampling_ratio/mean": 1.0673742294311523, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.817643642425537, "clip_ratio/low_mean": 0.12432807870209217, "clip_ratio/low_min": 0.12432807870209217, "clip_ratio/high_mean": 0.03415032755583525, "clip_ratio/high_max": 0.03415032755583525, "clip_ratio/region_mean": 0.15847840625792742, "reward_total_mean": 0.6222023367881775, "reward_meter_mean": 0.9594722390174866, "reward_meter_std": 0.06740762293338776, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9954468607902527, "reward_repeat_soft_std": 0.00810985453426838, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.1440672129392624, "reward_total_composite_mean": 0.6222023367881775, "reward_total_composite_std": 0.10977651923894882} {"timestamp_utc": "2026-04-13T07:54:33Z", "mode": "train", "global_step": 239, "epoch": 0.024008036162732296, "loss": 0.1266, "grad_norm": 16.885149002075195, "learning_rate": 9.27878787878788e-06, "num_tokens": 442355.0, "completions/mean_length": 42.625, "completions/min_length": 29.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.625, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.5577026605606079, "rewards/meter/std": 0.41305047273635864, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9940649271011353, "rewards/repeat_soft/std": 0.015189046040177345, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.4251362085342407, "rewards/total_composite/std": 0.199260875582695, "reward": 0.4251362085342407, "reward_std": 0.1992608904838562, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23422786593437195, "sampling/sampling_logp_difference/max": 1.3673014640808105, "sampling/importance_sampling_ratio/min": 0.25479358434677124, "sampling/importance_sampling_ratio/mean": 1.0716608762741089, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.5329028964042664, "clip_ratio/low_mean": 0.08765532821416855, "clip_ratio/low_min": 0.08765532821416855, "clip_ratio/high_mean": 0.08749033231288195, "clip_ratio/high_max": 0.08749033231288195, "clip_ratio/region_mean": 0.1751456605270505, "reward_total_mean": 0.4251362085342407, "reward_meter_mean": 0.5577026605606079, "reward_meter_std": 0.41305047273635864, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9940649271011353, "reward_repeat_soft_std": 0.015189046040177345, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.4251362085342407, "reward_total_composite_std": 0.199260875582695} {"timestamp_utc": "2026-04-13T07:54:41Z", "mode": "train", "global_step": 240, "epoch": 0.024108488196885988, "loss": 0.1303, "grad_norm": 11.940001487731934, "learning_rate": 9.275757575757577e-06, "num_tokens": 444059.0, "completions/mean_length": 56.0, "completions/min_length": 45.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.0, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.4955418109893799, "rewards/meter/std": 0.45711708068847656, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9967243671417236, "rewards/repeat_soft/std": 0.005859177093952894, "rewards/judge_quality/mean": 0.33125001192092896, "rewards/judge_quality/std": 0.13611313700675964, "rewards/total_composite/mean": 0.3668709099292755, "rewards/total_composite/std": 0.25199586153030396, "reward": 0.3668709099292755, "reward_std": 0.25199583172798157, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.26978856325149536, "sampling/sampling_logp_difference/max": 1.5231103897094727, "sampling/importance_sampling_ratio/min": 0.21803268790245056, "sampling/importance_sampling_ratio/mean": 1.0623533725738525, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.923114836215973, "clip_ratio/low_mean": 0.046531341038644314, "clip_ratio/low_min": 0.046531341038644314, "clip_ratio/high_mean": 0.12861622124910355, "clip_ratio/high_max": 0.12861622124910355, "clip_ratio/region_mean": 0.17514756228774786, "reward_total_mean": 0.3668709099292755, "reward_meter_mean": 0.4955418109893799, "reward_meter_std": 0.45711708068847656, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9967243671417236, "reward_repeat_soft_std": 0.005859177093952894, "reward_judge_quality_mean": 0.33125001192092896, "reward_judge_quality_std": 0.13611313700675964, "reward_total_composite_mean": 0.3668709099292755, "reward_total_composite_std": 0.25199586153030396} {"timestamp_utc": "2026-04-13T07:54:54Z", "mode": "train", "global_step": 241, "epoch": 0.02420894023103968, "loss": 0.0658, "grad_norm": 12.133151054382324, "learning_rate": 9.272727272727273e-06, "num_tokens": 445973.0, "completions/mean_length": 76.25, "completions/min_length": 54.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.25, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.19163626432418823, "rewards/meter/std": 0.26619383692741394, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9973886609077454, "rewards/repeat_soft/std": 0.0026149640325456858, "rewards/judge_quality/mean": 0.6025000214576721, "rewards/judge_quality/std": 0.35443115234375, "rewards/total_composite/mean": 0.41041338443756104, "rewards/total_composite/std": 0.22805547714233398, "reward": 0.41041338443756104, "reward_std": 0.22805547714233398, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22107474505901337, "sampling/sampling_logp_difference/max": 1.2895097732543945, "sampling/importance_sampling_ratio/min": 0.27540576457977295, "sampling/importance_sampling_ratio/mean": 1.0179635286331177, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.865700677037239, "clip_ratio/low_mean": 0.08136807568371296, "clip_ratio/low_min": 0.08136807568371296, "clip_ratio/high_mean": 0.09546291641891003, "clip_ratio/high_max": 0.09546291641891003, "clip_ratio/region_mean": 0.17683099210262299, "reward_total_mean": 0.41041338443756104, "reward_meter_mean": 0.19163626432418823, "reward_meter_std": 0.26619383692741394, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9973886609077454, "reward_repeat_soft_std": 0.0026149640325456858, "reward_judge_quality_mean": 0.6025000214576721, "reward_judge_quality_std": 0.35443115234375, "reward_total_composite_mean": 0.41041338443756104, "reward_total_composite_std": 0.22805547714233398} {"timestamp_utc": "2026-04-13T07:55:01Z", "mode": "train", "global_step": 242, "epoch": 0.02430939226519337, "loss": 0.0012, "grad_norm": 24.344709396362305, "learning_rate": 9.26969696969697e-06, "num_tokens": 447409.0, "completions/mean_length": 22.5, "completions/min_length": 20.0, "completions/max_length": 24.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.5, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 24.0, "rewards/meter/mean": 0.7496224641799927, "rewards/meter/std": 0.4358169138431549, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.0645727664232254, "rewards/total_composite/mean": 0.5515627264976501, "rewards/total_composite/std": 0.12039294838905334, "reward": 0.5515627264976501, "reward_std": 0.12039293348789215, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07915819436311722, "sampling/sampling_logp_difference/max": 0.7866744995117188, "sampling/importance_sampling_ratio/min": 0.45535656809806824, "sampling/importance_sampling_ratio/mean": 0.9973716139793396, "sampling/importance_sampling_ratio/max": 1.668638825416565, "entropy": 0.6392558962106705, "clip_ratio/low_mean": 0.07604166679084301, "clip_ratio/low_min": 0.07604166679084301, "clip_ratio/high_mean": 0.02766798483207822, "clip_ratio/high_max": 0.02766798483207822, "clip_ratio/region_mean": 0.10370965162292123, "reward_total_mean": 0.5515627264976501, "reward_meter_mean": 0.7496224641799927, "reward_meter_std": 0.4358169138431549, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.0645727664232254, "reward_total_composite_mean": 0.5515627264976501, "reward_total_composite_std": 0.12039294838905334} {"timestamp_utc": "2026-04-13T07:55:08Z", "mode": "train", "global_step": 243, "epoch": 0.02440984429934706, "loss": 0.1421, "grad_norm": 12.978425025939941, "learning_rate": 9.266666666666667e-06, "num_tokens": 449253.0, "completions/mean_length": 67.5, "completions/min_length": 44.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.5, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.4874016046524048, "rewards/meter/std": 0.4216694235801697, "rewards/count_adherence/mean": 0.8333333730697632, "rewards/count_adherence/std": 0.17817415297031403, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9921243190765381, "rewards/repeat_soft/std": 0.012025803327560425, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.4441221356391907, "rewards/total_composite/std": 0.13617344200611115, "reward": 0.4441221356391907, "reward_std": 0.13617344200611115, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2546895444393158, "sampling/sampling_logp_difference/max": 1.8767824172973633, "sampling/importance_sampling_ratio/min": 0.15308187901973724, "sampling/importance_sampling_ratio/mean": 1.0500438213348389, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.821613907814026, "clip_ratio/low_mean": 0.1031344085931778, "clip_ratio/low_min": 0.1031344085931778, "clip_ratio/high_mean": 0.10280172526836395, "clip_ratio/high_max": 0.10280172526836395, "clip_ratio/region_mean": 0.20593613386154175, "reward_total_mean": 0.4441221356391907, "reward_meter_mean": 0.4874016046524048, "reward_meter_std": 0.4216694235801697, "reward_count_adherence_mean": 0.8333333730697632, "reward_count_adherence_std": 0.17817415297031403, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9921243190765381, "reward_repeat_soft_std": 0.012025803327560425, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.4441221356391907, "reward_total_composite_std": 0.13617344200611115} {"timestamp_utc": "2026-04-13T07:55:16Z", "mode": "train", "global_step": 244, "epoch": 0.024510296333500752, "loss": -0.0164, "grad_norm": 10.075081825256348, "learning_rate": 9.263636363636364e-06, "num_tokens": 450906.0, "completions/mean_length": 58.625, "completions/min_length": 41.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.625, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.27877378463745117, "rewards/meter/std": 0.3891329765319824, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9942314624786377, "rewards/repeat_soft/std": 0.011857480742037296, "rewards/judge_quality/mean": 0.4337500333786011, "rewards/judge_quality/std": 0.15528200566768646, "rewards/total_composite/mean": 0.4079081416130066, "rewards/total_composite/std": 0.23522789776325226, "reward": 0.4079081416130066, "reward_std": 0.23522788286209106, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22137722373008728, "sampling/sampling_logp_difference/max": 1.596456527709961, "sampling/importance_sampling_ratio/min": 0.2026132047176361, "sampling/importance_sampling_ratio/mean": 1.0437744855880737, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6175760626792908, "clip_ratio/low_mean": 0.09715724922716618, "clip_ratio/low_min": 0.09715724922716618, "clip_ratio/high_mean": 0.05981331318616867, "clip_ratio/high_max": 0.05981331318616867, "clip_ratio/region_mean": 0.15697056241333485, "reward_total_mean": 0.4079081416130066, "reward_meter_mean": 0.27877378463745117, "reward_meter_std": 0.3891329765319824, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9942314624786377, "reward_repeat_soft_std": 0.011857480742037296, "reward_judge_quality_mean": 0.4337500333786011, "reward_judge_quality_std": 0.15528200566768646, "reward_total_composite_mean": 0.4079081416130066, "reward_total_composite_std": 0.23522789776325226} {"timestamp_utc": "2026-04-13T07:55:24Z", "mode": "train", "global_step": 245, "epoch": 0.024610748367654443, "loss": 0.0556, "grad_norm": 13.549051284790039, "learning_rate": 9.260606060606062e-06, "num_tokens": 452696.0, "completions/mean_length": 49.75, "completions/min_length": 36.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.75, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.7952672243118286, "rewards/meter/std": 0.2641472816467285, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9996022582054138, "rewards/repeat_soft/std": 0.001124941511079669, "rewards/judge_quality/mean": 0.3462499976158142, "rewards/judge_quality/std": 0.10336308926343918, "rewards/total_composite/mean": 0.4710841178894043, "rewards/total_composite/std": 0.20893138647079468, "reward": 0.4710841178894043, "reward_std": 0.20893137156963348, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22622895240783691, "sampling/sampling_logp_difference/max": 1.295872688293457, "sampling/importance_sampling_ratio/min": 0.2736589312553406, "sampling/importance_sampling_ratio/mean": 1.0102380514144897, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.124099910259247, "clip_ratio/low_mean": 0.043450853787362576, "clip_ratio/low_min": 0.043450853787362576, "clip_ratio/high_mean": 0.15173144079744816, "clip_ratio/high_max": 0.15173144079744816, "clip_ratio/region_mean": 0.19518229458481073, "reward_total_mean": 0.4710841178894043, "reward_meter_mean": 0.7952672243118286, "reward_meter_std": 0.2641472816467285, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9996022582054138, "reward_repeat_soft_std": 0.001124941511079669, "reward_judge_quality_mean": 0.3462499976158142, "reward_judge_quality_std": 0.10336308926343918, "reward_total_composite_mean": 0.4710841178894043, "reward_total_composite_std": 0.20893138647079468} {"timestamp_utc": "2026-04-13T07:55:34Z", "mode": "train", "global_step": 246, "epoch": 0.024711200401808138, "loss": -0.0181, "grad_norm": 5.99717378616333, "learning_rate": 9.257575757575759e-06, "num_tokens": 455847.0, "completions/mean_length": 197.875, "completions/min_length": 136.0, "completions/max_length": 296.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 197.875, "completions/min_terminated_length": 136.0, "completions/max_terminated_length": 296.0, "rewards/meter/mean": 0.19821536540985107, "rewards/meter/std": 0.17813539505004883, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.07715168595314026, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9967764616012573, "rewards/repeat_soft/std": 0.00359341804869473, "rewards/judge_quality/mean": 0.2137499898672104, "rewards/judge_quality/std": 0.10155048221349716, "rewards/total_composite/mean": 0.3095589280128479, "rewards/total_composite/std": 0.13021212816238403, "reward": 0.3095589280128479, "reward_std": 0.13021212816238403, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23955827951431274, "sampling/sampling_logp_difference/max": 1.8856773376464844, "sampling/importance_sampling_ratio/min": 0.15172624588012695, "sampling/importance_sampling_ratio/mean": 1.078747272491455, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 4.056632041931152, "clip_ratio/low_mean": 0.01977401040494442, "clip_ratio/low_min": 0.01977401040494442, "clip_ratio/high_mean": 0.15362998005002737, "clip_ratio/high_max": 0.15362998005002737, "clip_ratio/region_mean": 0.1734039904549718, "reward_total_mean": 0.3095589280128479, "reward_meter_mean": 0.19821536540985107, "reward_meter_std": 0.17813539505004883, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.07715168595314026, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9967764616012573, "reward_repeat_soft_std": 0.00359341804869473, "reward_judge_quality_mean": 0.2137499898672104, "reward_judge_quality_std": 0.10155048221349716, "reward_total_composite_mean": 0.3095589280128479, "reward_total_composite_std": 0.13021212816238403} {"timestamp_utc": "2026-04-13T07:55:43Z", "mode": "train", "global_step": 247, "epoch": 0.02481165243596183, "loss": -0.0468, "grad_norm": 12.217455863952637, "learning_rate": 9.254545454545454e-06, "num_tokens": 457542.0, "completions/mean_length": 51.875, "completions/min_length": 42.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.875, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.40306398272514343, "rewards/meter/std": 0.3892579972743988, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.997988760471344, "rewards/repeat_soft/std": 0.0030363963451236486, "rewards/judge_quality/mean": 0.38624998927116394, "rewards/judge_quality/std": 0.09545940905809402, "rewards/total_composite/mean": 0.2583388388156891, "rewards/total_composite/std": 0.2296733856201172, "reward": 0.2583388388156891, "reward_std": 0.22967340052127838, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24870580434799194, "sampling/sampling_logp_difference/max": 1.519944190979004, "sampling/importance_sampling_ratio/min": 0.2187241166830063, "sampling/importance_sampling_ratio/mean": 1.0517011880874634, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.1832044422626495, "clip_ratio/low_mean": 0.07400570623576641, "clip_ratio/low_min": 0.07400570623576641, "clip_ratio/high_mean": 0.1226517204195261, "clip_ratio/high_max": 0.1226517204195261, "clip_ratio/region_mean": 0.1966574266552925, "reward_total_mean": 0.2583388388156891, "reward_meter_mean": 0.40306398272514343, "reward_meter_std": 0.3892579972743988, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.997988760471344, "reward_repeat_soft_std": 0.0030363963451236486, "reward_judge_quality_mean": 0.38624998927116394, "reward_judge_quality_std": 0.09545940905809402, "reward_total_composite_mean": 0.2583388388156891, "reward_total_composite_std": 0.2296733856201172} {"timestamp_utc": "2026-04-13T07:55:55Z", "mode": "train", "global_step": 248, "epoch": 0.02491210447011552, "loss": -0.0577, "grad_norm": 5.521716117858887, "learning_rate": 9.251515151515152e-06, "num_tokens": 459522.0, "completions/mean_length": 151.5, "completions/min_length": 58.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 100.00000762939453, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 134.0, "rewards/meter/mean": 0.10642732679843903, "rewards/meter/std": 0.13759547472000122, "rewards/count_adherence/mean": 0.84375, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9973168969154358, "rewards/repeat_soft/std": 0.00279581593349576, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.21022097766399384, "rewards/total_composite/mean": 0.27132105827331543, "rewards/total_composite/std": 0.17161008715629578, "reward": 0.27132105827331543, "reward_std": 0.17161008715629578, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23343904316425323, "sampling/sampling_logp_difference/max": 1.4936189651489258, "sampling/importance_sampling_ratio/min": 0.2245585173368454, "sampling/importance_sampling_ratio/mean": 1.0569976568222046, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.9061760902404785, "clip_ratio/low_mean": 0.01772388070821762, "clip_ratio/low_min": 0.01772388070821762, "clip_ratio/high_mean": 0.15801917761564255, "clip_ratio/high_max": 0.15801917761564255, "clip_ratio/region_mean": 0.17574305832386017, "reward_total_mean": 0.27132105827331543, "reward_meter_mean": 0.10642732679843903, "reward_meter_std": 0.13759547472000122, "reward_count_adherence_mean": 0.84375, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9973168969154358, "reward_repeat_soft_std": 0.00279581593349576, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.21022097766399384, "reward_total_composite_mean": 0.27132105827331543, "reward_total_composite_std": 0.17161008715629578} {"timestamp_utc": "2026-04-13T07:56:01Z", "mode": "train", "global_step": 249, "epoch": 0.02501255650426921, "loss": 0.0666, "grad_norm": 23.50229263305664, "learning_rate": 9.248484848484849e-06, "num_tokens": 460936.0, "completions/mean_length": 31.75, "completions/min_length": 24.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.75, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.6344245076179504, "rewards/meter/std": 0.3527393937110901, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9875197410583496, "rewards/repeat_soft/std": 0.01588606648147106, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.5660109519958496, "rewards/total_composite/std": 0.13045436143875122, "reward": 0.5660109519958496, "reward_std": 0.13045434653759003, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18733884394168854, "sampling/sampling_logp_difference/max": 1.8012290000915527, "sampling/importance_sampling_ratio/min": 0.16509586572647095, "sampling/importance_sampling_ratio/mean": 1.0040596723556519, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1648800373077393, "clip_ratio/low_mean": 0.06251262687146664, "clip_ratio/low_min": 0.06251262687146664, "clip_ratio/high_mean": 0.11953509598970413, "clip_ratio/high_max": 0.11953509598970413, "clip_ratio/region_mean": 0.18204772286117077, "reward_total_mean": 0.5660109519958496, "reward_meter_mean": 0.6344245076179504, "reward_meter_std": 0.3527393937110901, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9875197410583496, "reward_repeat_soft_std": 0.01588606648147106, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.5660109519958496, "reward_total_composite_std": 0.13045436143875122} {"timestamp_utc": "2026-04-13T07:56:28Z", "mode": "train", "global_step": 250, "epoch": 0.025113008538422903, "loss": -0.2321, "grad_norm": 2.7613444328308105, "learning_rate": 9.245454545454546e-06, "num_tokens": 463523.0, "completions/mean_length": 309.375, "completions/min_length": 160.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 187.8000030517578, "completions/min_terminated_length": 160.0, "completions/max_terminated_length": 234.0, "rewards/meter/mean": 0.1741102635860443, "rewards/meter/std": 0.27033716440200806, "rewards/count_adherence/mean": 0.7000000476837158, "rewards/count_adherence/std": 0.1511857807636261, "rewards/hard_gate/mean": 0.5, "rewards/hard_gate/std": 0.5345224738121033, "rewards/repeat_soft/mean": 0.999889612197876, "rewards/repeat_soft/std": 0.00021023683075327426, "rewards/judge_quality/mean": 0.11749999970197678, "rewards/judge_quality/std": 0.06902380287647247, "rewards/total_composite/mean": 0.16109412908554077, "rewards/total_composite/std": 0.17332017421722412, "reward": 0.16109412908554077, "reward_std": 0.17332017421722412, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24947352707386017, "sampling/sampling_logp_difference/max": 1.2111396789550781, "sampling/importance_sampling_ratio/min": 0.2978576123714447, "sampling/importance_sampling_ratio/mean": 1.0666618347167969, "sampling/importance_sampling_ratio/max": 1.9700634479522705, "entropy": 2.7018612027168274, "clip_ratio/low_mean": 0.01875000074505806, "clip_ratio/low_min": 0.01875000074505806, "clip_ratio/high_mean": 0.08803261630237103, "clip_ratio/high_max": 0.08803261630237103, "clip_ratio/region_mean": 0.10678261704742908, "reward_total_mean": 0.16109412908554077, "reward_meter_mean": 0.1741102635860443, "reward_meter_std": 0.27033716440200806, "reward_count_adherence_mean": 0.7000000476837158, "reward_count_adherence_std": 0.1511857807636261, "reward_hard_gate_mean": 0.5, "reward_hard_gate_std": 0.5345224738121033, "reward_repeat_soft_mean": 0.999889612197876, "reward_repeat_soft_std": 0.00021023683075327426, "reward_judge_quality_mean": 0.11749999970197678, "reward_judge_quality_std": 0.06902380287647247, "reward_total_composite_mean": 0.16109412908554077, "reward_total_composite_std": 0.17332017421722412} {"timestamp_utc": "2026-04-13T07:57:42Z", "mode": "eval", "global_step": 250, "epoch": 0.025113008538422903, "eval_loss": NaN, "eval_runtime": 74.0879, "eval_samples_per_second": 1.08, "eval_steps_per_second": 0.135, "eval_num_tokens": 463523.0, "eval_completions/mean_length": 105.1875, "eval_completions/min_length": 33.0, "eval_completions/max_length": 279.8, "eval_completions/clipped_ratio": 0.0375, "eval_completions/mean_terminated_length": 89.50357284545899, "eval_completions/min_terminated_length": 33.0, "eval_completions/max_terminated_length": 187.5, "eval_rewards/meter/mean": 0.4927772879600525, "eval_rewards/meter/std": 0.36216977536678313, "eval_rewards/count_adherence/mean": 0.9197916686534882, "eval_rewards/count_adherence/std": 0.15851535201072692, "eval_rewards/hard_gate/mean": 0.9125, "eval_rewards/hard_gate/std": 0.2230676978826523, "eval_rewards/repeat_soft/mean": 0.9888128161430358, "eval_rewards/repeat_soft/std": 0.02121427790261805, "eval_rewards/judge_quality/mean": 0.3487500011920929, "eval_rewards/judge_quality/std": 0.1130803014151752, "eval_rewards/total_composite/mean": 0.41938657462596896, "eval_rewards/total_composite/std": 0.14896008148789405, "eval_reward": 0.41938657462596896, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.18254771083593369, "eval_sampling/sampling_logp_difference/max": 1.249609899520874, "eval_sampling/importance_sampling_ratio/min": 0.29058967232704164, "eval_sampling/importance_sampling_ratio/mean": 1.0555893182754517, "eval_sampling/importance_sampling_ratio/max": 1.7040584444999696, "eval_entropy": 3.2519503593444825, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.41938657462596896, "eval_reward_meter_mean": 0.4927772879600525, "eval_reward_meter_std": 0.36216977536678313, "eval_reward_count_adherence_mean": 0.9197916686534882, "eval_reward_count_adherence_std": 0.15851535201072692, "eval_reward_hard_gate_mean": 0.9125, "eval_reward_hard_gate_std": 0.2230676978826523, "eval_reward_repeat_soft_mean": 0.9888128161430358, "eval_reward_repeat_soft_std": 0.02121427790261805, "eval_reward_judge_quality_mean": 0.3487500011920929, "eval_reward_judge_quality_std": 0.1130803014151752, "eval_reward_total_composite_mean": 0.41938657462596896, "eval_reward_total_composite_std": 0.14896008148789405} {"timestamp_utc": "2026-04-13T07:57:52Z", "mode": "train", "global_step": 251, "epoch": 0.025213460572576594, "loss": 0.0505, "grad_norm": 15.902443885803223, "learning_rate": 9.242424242424244e-06, "num_tokens": 465093.0, "completions/mean_length": 40.25, "completions/min_length": 32.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.25, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.18963100016117096, "rewards/meter/std": 0.2852773070335388, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9908022880554199, "rewards/repeat_soft/std": 0.007465601898729801, "rewards/judge_quality/mean": 0.48624998331069946, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.40558940172195435, "rewards/total_composite/std": 0.0778096541762352, "reward": 0.40558940172195435, "reward_std": 0.0778096541762352, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22490194439888, "sampling/sampling_logp_difference/max": 1.2477350234985352, "sampling/importance_sampling_ratio/min": 0.2871544659137726, "sampling/importance_sampling_ratio/mean": 1.0595271587371826, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.8220799267292023, "clip_ratio/low_mean": 0.12860641907900572, "clip_ratio/low_min": 0.12860641907900572, "clip_ratio/high_mean": 0.031746032647788525, "clip_ratio/high_max": 0.031746032647788525, "clip_ratio/region_mean": 0.16035245172679424, "reward_total_mean": 0.40558940172195435, "reward_meter_mean": 0.18963100016117096, "reward_meter_std": 0.2852773070335388, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9908022880554199, "reward_repeat_soft_std": 0.007465601898729801, "reward_judge_quality_mean": 0.48624998331069946, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.40558940172195435, "reward_total_composite_std": 0.0778096541762352} {"timestamp_utc": "2026-04-13T07:58:01Z", "mode": "train", "global_step": 252, "epoch": 0.025313912606730285, "loss": 0.0901, "grad_norm": 11.631461143493652, "learning_rate": 9.23939393939394e-06, "num_tokens": 466985.0, "completions/mean_length": 68.5, "completions/min_length": 47.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.5, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.23365449905395508, "rewards/meter/std": 0.22981607913970947, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9940780401229858, "rewards/repeat_soft/std": 0.012517235241830349, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.40456604957580566, "rewards/total_composite/std": 0.0727265328168869, "reward": 0.40456604957580566, "reward_std": 0.0727265328168869, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2311936616897583, "sampling/sampling_logp_difference/max": 1.5864734649658203, "sampling/importance_sampling_ratio/min": 0.2046460211277008, "sampling/importance_sampling_ratio/mean": 1.0510509014129639, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.064756900072098, "clip_ratio/low_mean": 0.11715154815465212, "clip_ratio/low_min": 0.11715154815465212, "clip_ratio/high_mean": 0.06316489353775978, "clip_ratio/high_max": 0.06316489353775978, "clip_ratio/region_mean": 0.1803164416924119, "reward_total_mean": 0.40456604957580566, "reward_meter_mean": 0.23365449905395508, "reward_meter_std": 0.22981607913970947, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9940780401229858, "reward_repeat_soft_std": 0.012517235241830349, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.40456604957580566, "reward_total_composite_std": 0.0727265328168869} {"timestamp_utc": "2026-04-13T07:58:08Z", "mode": "train", "global_step": 253, "epoch": 0.02541436464088398, "loss": 0.2207, "grad_norm": 14.660782814025879, "learning_rate": 9.236363636363636e-06, "num_tokens": 468575.0, "completions/mean_length": 46.75, "completions/min_length": 26.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.75, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.6230303049087524, "rewards/meter/std": 0.45745065808296204, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9883996844291687, "rewards/repeat_soft/std": 0.020203450694680214, "rewards/judge_quality/mean": 0.5275000333786011, "rewards/judge_quality/std": 0.2499571591615677, "rewards/total_composite/mean": 0.5688772201538086, "rewards/total_composite/std": 0.24092671275138855, "reward": 0.5688772201538086, "reward_std": 0.24092671275138855, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21159371733665466, "sampling/sampling_logp_difference/max": 1.2312355041503906, "sampling/importance_sampling_ratio/min": 0.2919316589832306, "sampling/importance_sampling_ratio/mean": 1.074669599533081, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.669960081577301, "clip_ratio/low_mean": 0.08137993142008781, "clip_ratio/low_min": 0.08137993142008781, "clip_ratio/high_mean": 0.0769733702763915, "clip_ratio/high_max": 0.0769733702763915, "clip_ratio/region_mean": 0.15835330169647932, "reward_total_mean": 0.5688772201538086, "reward_meter_mean": 0.6230303049087524, "reward_meter_std": 0.45745065808296204, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9883996844291687, "reward_repeat_soft_std": 0.020203450694680214, "reward_judge_quality_mean": 0.5275000333786011, "reward_judge_quality_std": 0.2499571591615677, "reward_total_composite_mean": 0.5688772201538086, "reward_total_composite_std": 0.24092671275138855} {"timestamp_utc": "2026-04-13T07:58:28Z", "mode": "train", "global_step": 254, "epoch": 0.02551481667503767, "loss": 0.0185, "grad_norm": 16.987140655517578, "learning_rate": 9.233333333333334e-06, "num_tokens": 470489.0, "completions/mean_length": 71.25, "completions/min_length": 57.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.25, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.5752930641174316, "rewards/meter/std": 0.3816414475440979, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9927878975868225, "rewards/repeat_soft/std": 0.007032928988337517, "rewards/judge_quality/mean": 0.29249998927116394, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.46241429448127747, "rewards/total_composite/std": 0.10644890367984772, "reward": 0.46241429448127747, "reward_std": 0.10644891113042831, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22552765905857086, "sampling/sampling_logp_difference/max": 2.155546188354492, "sampling/importance_sampling_ratio/min": 0.11583990603685379, "sampling/importance_sampling_ratio/mean": 1.0416035652160645, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.875844269990921, "clip_ratio/low_mean": 0.11048714444041252, "clip_ratio/low_min": 0.11048714444041252, "clip_ratio/high_mean": 0.10103101842105389, "clip_ratio/high_max": 0.10103101842105389, "clip_ratio/region_mean": 0.2115181628614664, "reward_total_mean": 0.46241429448127747, "reward_meter_mean": 0.5752930641174316, "reward_meter_std": 0.3816414475440979, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9927878975868225, "reward_repeat_soft_std": 0.007032928988337517, "reward_judge_quality_mean": 0.29249998927116394, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.46241429448127747, "reward_total_composite_std": 0.10644890367984772} {"timestamp_utc": "2026-04-13T07:58:36Z", "mode": "train", "global_step": 255, "epoch": 0.025615268709191362, "loss": -0.0058, "grad_norm": 9.851515769958496, "learning_rate": 9.23030303030303e-06, "num_tokens": 472370.0, "completions/mean_length": 61.125, "completions/min_length": 34.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.125, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.6299946308135986, "rewards/meter/std": 0.4327425956726074, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.24800792336463928, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9937536120414734, "rewards/repeat_soft/std": 0.012771286070346832, "rewards/judge_quality/mean": 0.3387500047683716, "rewards/judge_quality/std": 0.09538455307483673, "rewards/total_composite/mean": 0.4724371135234833, "rewards/total_composite/std": 0.14866700768470764, "reward": 0.4724371135234833, "reward_std": 0.14866699278354645, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23003731667995453, "sampling/sampling_logp_difference/max": 1.5137968063354492, "sampling/importance_sampling_ratio/min": 0.22007280588150024, "sampling/importance_sampling_ratio/mean": 1.0665873289108276, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.4186623990535736, "clip_ratio/low_mean": 0.12104959040880203, "clip_ratio/low_min": 0.12104959040880203, "clip_ratio/high_mean": 0.1058238660916686, "clip_ratio/high_max": 0.1058238660916686, "clip_ratio/region_mean": 0.22687345650047064, "reward_total_mean": 0.4724371135234833, "reward_meter_mean": 0.6299946308135986, "reward_meter_std": 0.4327425956726074, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.24800792336463928, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9937536120414734, "reward_repeat_soft_std": 0.012771286070346832, "reward_judge_quality_mean": 0.3387500047683716, "reward_judge_quality_std": 0.09538455307483673, "reward_total_composite_mean": 0.4724371135234833, "reward_total_composite_std": 0.14866700768470764} {"timestamp_utc": "2026-04-13T07:58:43Z", "mode": "train", "global_step": 256, "epoch": 0.025715720743345053, "loss": 0.0585, "grad_norm": 13.359979629516602, "learning_rate": 9.227272727272728e-06, "num_tokens": 474129.0, "completions/mean_length": 46.875, "completions/min_length": 40.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.875, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.7647510170936584, "rewards/meter/std": 0.263922780752182, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.991895854473114, "rewards/repeat_soft/std": 0.01966020278632641, "rewards/judge_quality/mean": 0.48499998450279236, "rewards/judge_quality/std": 0.15937379002571106, "rewards/total_composite/mean": 0.5823091864585876, "rewards/total_composite/std": 0.11638780683279037, "reward": 0.5823091864585876, "reward_std": 0.11638780683279037, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2153816670179367, "sampling/sampling_logp_difference/max": 1.2624378204345703, "sampling/importance_sampling_ratio/min": 0.282963365316391, "sampling/importance_sampling_ratio/mean": 1.0510693788528442, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.654299169778824, "clip_ratio/low_mean": 0.08597616292536259, "clip_ratio/low_min": 0.08597616292536259, "clip_ratio/high_mean": 0.10753452405333519, "clip_ratio/high_max": 0.10753452405333519, "clip_ratio/region_mean": 0.19351068697869778, "reward_total_mean": 0.5823091864585876, "reward_meter_mean": 0.7647510170936584, "reward_meter_std": 0.263922780752182, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.991895854473114, "reward_repeat_soft_std": 0.01966020278632641, "reward_judge_quality_mean": 0.48499998450279236, "reward_judge_quality_std": 0.15937379002571106, "reward_total_composite_mean": 0.5823091864585876, "reward_total_composite_std": 0.11638780683279037} {"timestamp_utc": "2026-04-13T07:58:53Z", "mode": "train", "global_step": 257, "epoch": 0.025816172777498744, "loss": 0.0172, "grad_norm": 5.30784273147583, "learning_rate": 9.224242424242424e-06, "num_tokens": 477513.0, "completions/mean_length": 210.0, "completions/min_length": 158.0, "completions/max_length": 399.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 210.0, "completions/min_terminated_length": 158.0, "completions/max_terminated_length": 399.0, "rewards/meter/mean": 0.6121722459793091, "rewards/meter/std": 0.3405526876449585, "rewards/count_adherence/mean": 0.8500000238418579, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9984015226364136, "rewards/repeat_soft/std": 0.0018976051360368729, "rewards/judge_quality/mean": 0.17249999940395355, "rewards/judge_quality/std": 0.12623673677444458, "rewards/total_composite/mean": 0.35219359397888184, "rewards/total_composite/std": 0.1643473207950592, "reward": 0.35219359397888184, "reward_std": 0.164347305893898, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23884940147399902, "sampling/sampling_logp_difference/max": 1.5050020217895508, "sampling/importance_sampling_ratio/min": 0.2220168560743332, "sampling/importance_sampling_ratio/mean": 1.0711714029312134, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 4.338009357452393, "clip_ratio/low_mean": 0.0566375432536006, "clip_ratio/low_min": 0.0566375432536006, "clip_ratio/high_mean": 0.09318997152149677, "clip_ratio/high_max": 0.09318997152149677, "clip_ratio/region_mean": 0.14982751477509737, "reward_total_mean": 0.35219359397888184, "reward_meter_mean": 0.6121722459793091, "reward_meter_std": 0.3405526876449585, "reward_count_adherence_mean": 0.8500000238418579, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9984015226364136, "reward_repeat_soft_std": 0.0018976051360368729, "reward_judge_quality_mean": 0.17249999940395355, "reward_judge_quality_std": 0.12623673677444458, "reward_total_composite_mean": 0.35219359397888184, "reward_total_composite_std": 0.1643473207950592} {"timestamp_utc": "2026-04-13T07:59:01Z", "mode": "train", "global_step": 258, "epoch": 0.025916624811652435, "loss": 0.0871, "grad_norm": 11.726977348327637, "learning_rate": 9.221212121212123e-06, "num_tokens": 479486.0, "completions/mean_length": 87.625, "completions/min_length": 50.0, "completions/max_length": 121.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.625, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.5028528571128845, "rewards/meter/std": 0.2695654034614563, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9962344169616699, "rewards/repeat_soft/std": 0.004130613524466753, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.49441421031951904, "rewards/total_composite/std": 0.08856210857629776, "reward": 0.49441421031951904, "reward_std": 0.08856210857629776, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22967101633548737, "sampling/sampling_logp_difference/max": 1.4539060592651367, "sampling/importance_sampling_ratio/min": 0.23365582525730133, "sampling/importance_sampling_ratio/mean": 1.0647671222686768, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.430586889386177, "clip_ratio/low_mean": 0.10441025160253048, "clip_ratio/low_min": 0.10441025160253048, "clip_ratio/high_mean": 0.09899860806763172, "clip_ratio/high_max": 0.09899860806763172, "clip_ratio/region_mean": 0.2034088596701622, "reward_total_mean": 0.49441421031951904, "reward_meter_mean": 0.5028528571128845, "reward_meter_std": 0.2695654034614563, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9962344169616699, "reward_repeat_soft_std": 0.004130613524466753, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.49441421031951904, "reward_total_composite_std": 0.08856210857629776} {"timestamp_utc": "2026-04-13T07:59:13Z", "mode": "train", "global_step": 259, "epoch": 0.026017076845806127, "loss": -0.0702, "grad_norm": 1.9631386995315552, "learning_rate": 9.21818181818182e-06, "num_tokens": 480908.0, "completions/mean_length": 85.75, "completions/min_length": 17.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 24.85714340209961, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.6726659536361694, "rewards/meter/std": 0.3912229835987091, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9622172117233276, "rewards/repeat_soft/std": 0.0007998839719220996, "rewards/judge_quality/mean": 0.32625001668930054, "rewards/judge_quality/std": 0.15592464804649353, "rewards/total_composite/mean": 0.44956642389297485, "rewards/total_composite/std": 0.20386633276939392, "reward": 0.44956642389297485, "reward_std": 0.20386633276939392, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1834326833486557, "sampling/sampling_logp_difference/max": 0.8966245651245117, "sampling/importance_sampling_ratio/min": 0.40794432163238525, "sampling/importance_sampling_ratio/mean": 1.0076311826705933, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7231449484825134, "clip_ratio/low_mean": 0.04589160904288292, "clip_ratio/low_min": 0.04589160904288292, "clip_ratio/high_mean": 0.13169217109680176, "clip_ratio/high_max": 0.13169217109680176, "clip_ratio/region_mean": 0.17758378013968468, "reward_total_mean": 0.44956642389297485, "reward_meter_mean": 0.6726659536361694, "reward_meter_std": 0.3912229835987091, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9622172117233276, "reward_repeat_soft_std": 0.0007998839719220996, "reward_judge_quality_mean": 0.32625001668930054, "reward_judge_quality_std": 0.15592464804649353, "reward_total_composite_mean": 0.44956642389297485, "reward_total_composite_std": 0.20386633276939392} {"timestamp_utc": "2026-04-13T07:59:20Z", "mode": "train", "global_step": 260, "epoch": 0.026117528879959818, "loss": -0.0215, "grad_norm": 12.083062171936035, "learning_rate": 9.215151515151515e-06, "num_tokens": 482528.0, "completions/mean_length": 46.5, "completions/min_length": 37.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.5, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.44499480724334717, "rewards/meter/std": 0.3269158899784088, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9971332550048828, "rewards/repeat_soft/std": 0.0036495414096862078, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.1940544992685318, "rewards/total_composite/mean": 0.47453099489212036, "rewards/total_composite/std": 0.08867525309324265, "reward": 0.47453099489212036, "reward_std": 0.08867524564266205, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21196623146533966, "sampling/sampling_logp_difference/max": 1.0982379913330078, "sampling/importance_sampling_ratio/min": 0.3334581255912781, "sampling/importance_sampling_ratio/mean": 1.0493136644363403, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3388479202985764, "clip_ratio/low_mean": 0.08928120322525501, "clip_ratio/low_min": 0.08928120322525501, "clip_ratio/high_mean": 0.08731626439839602, "clip_ratio/high_max": 0.08731626439839602, "clip_ratio/region_mean": 0.17659746762365103, "reward_total_mean": 0.47453099489212036, "reward_meter_mean": 0.44499480724334717, "reward_meter_std": 0.3269158899784088, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9971332550048828, "reward_repeat_soft_std": 0.0036495414096862078, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.1940544992685318, "reward_total_composite_mean": 0.47453099489212036, "reward_total_composite_std": 0.08867525309324265} {"timestamp_utc": "2026-04-13T07:59:28Z", "mode": "train", "global_step": 261, "epoch": 0.026217980914113512, "loss": 0.066, "grad_norm": 13.19766616821289, "learning_rate": 9.212121212121213e-06, "num_tokens": 484183.0, "completions/mean_length": 48.875, "completions/min_length": 40.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.875, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.3864549398422241, "rewards/meter/std": 0.3447204530239105, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9985442757606506, "rewards/repeat_soft/std": 0.00270977895706892, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.1529705971479416, "rewards/total_composite/mean": 0.32365351915359497, "rewards/total_composite/std": 0.21631422638893127, "reward": 0.32365351915359497, "reward_std": 0.21631422638893127, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23518016934394836, "sampling/sampling_logp_difference/max": 1.751021385192871, "sampling/importance_sampling_ratio/min": 0.17359653115272522, "sampling/importance_sampling_ratio/mean": 1.0690120458602905, "sampling/importance_sampling_ratio/max": 1.978293776512146, "entropy": 3.245684951543808, "clip_ratio/low_mean": 0.0569117646664381, "clip_ratio/low_min": 0.0569117646664381, "clip_ratio/high_mean": 0.14734195545315742, "clip_ratio/high_max": 0.14734195545315742, "clip_ratio/region_mean": 0.20425372011959553, "reward_total_mean": 0.32365351915359497, "reward_meter_mean": 0.3864549398422241, "reward_meter_std": 0.3447204530239105, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9985442757606506, "reward_repeat_soft_std": 0.00270977895706892, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.1529705971479416, "reward_total_composite_mean": 0.32365351915359497, "reward_total_composite_std": 0.21631422638893127} {"timestamp_utc": "2026-04-13T07:59:34Z", "mode": "train", "global_step": 262, "epoch": 0.026318432948267204, "loss": 0.0519, "grad_norm": 16.314903259277344, "learning_rate": 9.20909090909091e-06, "num_tokens": 485761.0, "completions/mean_length": 42.25, "completions/min_length": 36.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.25, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.7184869647026062, "rewards/meter/std": 0.397292822599411, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9871054887771606, "rewards/repeat_soft/std": 0.02460484392940998, "rewards/judge_quality/mean": 0.38624998927116394, "rewards/judge_quality/std": 0.09545940905809402, "rewards/total_composite/mean": 0.5435694456100464, "rewards/total_composite/std": 0.10916414856910706, "reward": 0.5435694456100464, "reward_std": 0.10916414856910706, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2332073450088501, "sampling/sampling_logp_difference/max": 1.291060447692871, "sampling/importance_sampling_ratio/min": 0.27497902512550354, "sampling/importance_sampling_ratio/mean": 1.053661584854126, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.5997531712055206, "clip_ratio/low_mean": 0.037946428172290325, "clip_ratio/low_min": 0.037946428172290325, "clip_ratio/high_mean": 0.14962166547775269, "clip_ratio/high_max": 0.14962166547775269, "clip_ratio/region_mean": 0.187568093650043, "reward_total_mean": 0.5435694456100464, "reward_meter_mean": 0.7184869647026062, "reward_meter_std": 0.397292822599411, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9871054887771606, "reward_repeat_soft_std": 0.02460484392940998, "reward_judge_quality_mean": 0.38624998927116394, "reward_judge_quality_std": 0.09545940905809402, "reward_total_composite_mean": 0.5435694456100464, "reward_total_composite_std": 0.10916414856910706} {"timestamp_utc": "2026-04-13T07:59:41Z", "mode": "train", "global_step": 263, "epoch": 0.026418884982420895, "loss": -0.032, "grad_norm": 10.155993461608887, "learning_rate": 9.206060606060607e-06, "num_tokens": 487787.0, "completions/mean_length": 81.25, "completions/min_length": 66.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 81.25, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.5438829660415649, "rewards/meter/std": 0.40771350264549255, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.984946608543396, "rewards/repeat_soft/std": 0.015459898859262466, "rewards/judge_quality/mean": 0.41499999165534973, "rewards/judge_quality/std": 0.14520922303199768, "rewards/total_composite/mean": 0.4312058687210083, "rewards/total_composite/std": 0.19852708280086517, "reward": 0.4312058687210083, "reward_std": 0.19852708280086517, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2361883819103241, "sampling/sampling_logp_difference/max": 1.3174123764038086, "sampling/importance_sampling_ratio/min": 0.26782742142677307, "sampling/importance_sampling_ratio/mean": 1.0664438009262085, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.335661917924881, "clip_ratio/low_mean": 0.07798272278159857, "clip_ratio/low_min": 0.07798272278159857, "clip_ratio/high_mean": 0.0985830519348383, "clip_ratio/high_max": 0.0985830519348383, "clip_ratio/region_mean": 0.17656577471643686, "reward_total_mean": 0.4312058687210083, "reward_meter_mean": 0.5438829660415649, "reward_meter_std": 0.40771350264549255, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.984946608543396, "reward_repeat_soft_std": 0.015459898859262466, "reward_judge_quality_mean": 0.41499999165534973, "reward_judge_quality_std": 0.14520922303199768, "reward_total_composite_mean": 0.4312058687210083, "reward_total_composite_std": 0.19852708280086517} {"timestamp_utc": "2026-04-13T07:59:48Z", "mode": "train", "global_step": 264, "epoch": 0.026519337016574586, "loss": 0.0807, "grad_norm": 9.939325332641602, "learning_rate": 9.203030303030304e-06, "num_tokens": 489954.0, "completions/mean_length": 95.875, "completions/min_length": 68.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.875, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.39289143681526184, "rewards/meter/std": 0.3128518760204315, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9843380451202393, "rewards/repeat_soft/std": 0.029751332476735115, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.44051191210746765, "rewards/total_composite/std": 0.0717383399605751, "reward": 0.44051191210746765, "reward_std": 0.0717383399605751, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21802948415279388, "sampling/sampling_logp_difference/max": 1.4610528945922852, "sampling/importance_sampling_ratio/min": 0.23199188709259033, "sampling/importance_sampling_ratio/mean": 1.0672013759613037, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.0699095129966736, "clip_ratio/low_mean": 0.07578411884605885, "clip_ratio/low_min": 0.07578411884605885, "clip_ratio/high_mean": 0.13615627773106098, "clip_ratio/high_max": 0.13615627773106098, "clip_ratio/region_mean": 0.21194039657711983, "reward_total_mean": 0.44051191210746765, "reward_meter_mean": 0.39289143681526184, "reward_meter_std": 0.3128518760204315, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9843380451202393, "reward_repeat_soft_std": 0.029751332476735115, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.44051191210746765, "reward_total_composite_std": 0.0717383399605751} {"timestamp_utc": "2026-04-13T07:59:56Z", "mode": "train", "global_step": 265, "epoch": 0.026619789050728277, "loss": 0.0817, "grad_norm": 13.237419128417969, "learning_rate": 9.200000000000002e-06, "num_tokens": 492168.0, "completions/mean_length": 99.75, "completions/min_length": 83.0, "completions/max_length": 123.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 99.75, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.8052846193313599, "rewards/meter/std": 0.24584336578845978, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9940801858901978, "rewards/repeat_soft/std": 0.0032723494805395603, "rewards/judge_quality/mean": 0.5074999928474426, "rewards/judge_quality/std": 0.1642080694437027, "rewards/total_composite/mean": 0.6095613837242126, "rewards/total_composite/std": 0.12963679432868958, "reward": 0.6095613837242126, "reward_std": 0.12963679432868958, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20225085318088531, "sampling/sampling_logp_difference/max": 2.628049373626709, "sampling/importance_sampling_ratio/min": 0.07221920043230057, "sampling/importance_sampling_ratio/mean": 1.0143296718597412, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3046093508601189, "clip_ratio/low_mean": 0.0953658726066351, "clip_ratio/low_min": 0.0953658726066351, "clip_ratio/high_mean": 0.07775945216417313, "clip_ratio/high_max": 0.07775945216417313, "clip_ratio/region_mean": 0.17312532477080822, "reward_total_mean": 0.6095613837242126, "reward_meter_mean": 0.8052846193313599, "reward_meter_std": 0.24584336578845978, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9940801858901978, "reward_repeat_soft_std": 0.0032723494805395603, "reward_judge_quality_mean": 0.5074999928474426, "reward_judge_quality_std": 0.1642080694437027, "reward_total_composite_mean": 0.6095613837242126, "reward_total_composite_std": 0.12963679432868958} {"timestamp_utc": "2026-04-13T08:00:03Z", "mode": "train", "global_step": 266, "epoch": 0.026720241084881968, "loss": 0.1828, "grad_norm": 18.29632568359375, "learning_rate": 9.196969696969697e-06, "num_tokens": 493579.0, "completions/mean_length": 26.375, "completions/min_length": 20.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.375, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.8836090564727783, "rewards/meter/std": 0.26969626545906067, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.3999999761581421, "rewards/judge_quality/std": 0.09258200973272324, "rewards/total_composite/mean": 0.583101749420166, "rewards/total_composite/std": 0.09299547225236893, "reward": 0.583101749420166, "reward_std": 0.09299547970294952, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23211590945720673, "sampling/sampling_logp_difference/max": 1.0317764282226562, "sampling/importance_sampling_ratio/min": 0.3563733398914337, "sampling/importance_sampling_ratio/mean": 1.0721259117126465, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.2353347837924957, "clip_ratio/low_mean": 0.050245098769664764, "clip_ratio/low_min": 0.050245098769664764, "clip_ratio/high_mean": 0.17944775335490704, "clip_ratio/high_max": 0.17944775335490704, "clip_ratio/region_mean": 0.2296928521245718, "reward_total_mean": 0.583101749420166, "reward_meter_mean": 0.8836090564727783, "reward_meter_std": 0.26969626545906067, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.3999999761581421, "reward_judge_quality_std": 0.09258200973272324, "reward_total_composite_mean": 0.583101749420166, "reward_total_composite_std": 0.09299547225236893} {"timestamp_utc": "2026-04-13T08:00:11Z", "mode": "train", "global_step": 267, "epoch": 0.02682069311903566, "loss": 0.1007, "grad_norm": 11.504515647888184, "learning_rate": 9.193939393939395e-06, "num_tokens": 495297.0, "completions/mean_length": 54.75, "completions/min_length": 43.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.75, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.24899473786354065, "rewards/meter/std": 0.3889537453651428, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9957927465438843, "rewards/repeat_soft/std": 0.00977754034101963, "rewards/judge_quality/mean": 0.3675000071525574, "rewards/judge_quality/std": 0.18100908398628235, "rewards/total_composite/mean": 0.4044429063796997, "rewards/total_composite/std": 0.09395794570446014, "reward": 0.4044429063796997, "reward_std": 0.09395796060562134, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23803894221782684, "sampling/sampling_logp_difference/max": 1.2172846794128418, "sampling/importance_sampling_ratio/min": 0.31349968910217285, "sampling/importance_sampling_ratio/mean": 1.0595011711120605, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.626153290271759, "clip_ratio/low_mean": 0.11126524023711681, "clip_ratio/low_min": 0.11126524023711681, "clip_ratio/high_mean": 0.044706499204039574, "clip_ratio/high_max": 0.044706499204039574, "clip_ratio/region_mean": 0.1559717394411564, "reward_total_mean": 0.4044429063796997, "reward_meter_mean": 0.24899473786354065, "reward_meter_std": 0.3889537453651428, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9957927465438843, "reward_repeat_soft_std": 0.00977754034101963, "reward_judge_quality_mean": 0.3675000071525574, "reward_judge_quality_std": 0.18100908398628235, "reward_total_composite_mean": 0.4044429063796997, "reward_total_composite_std": 0.09395794570446014} {"timestamp_utc": "2026-04-13T08:00:19Z", "mode": "train", "global_step": 268, "epoch": 0.02692114515318935, "loss": 0.1384, "grad_norm": 18.148683547973633, "learning_rate": 9.190909090909092e-06, "num_tokens": 496902.0, "completions/mean_length": 33.625, "completions/min_length": 28.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.625, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.7711975574493408, "rewards/meter/std": 0.2813066840171814, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9876835346221924, "rewards/repeat_soft/std": 0.009118892252445221, "rewards/judge_quality/mean": 0.7737500071525574, "rewards/judge_quality/std": 0.22032040357589722, "rewards/total_composite/mean": 0.7391894459724426, "rewards/total_composite/std": 0.189705029129982, "reward": 0.7391894459724426, "reward_std": 0.189705029129982, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11177510023117065, "sampling/sampling_logp_difference/max": 1.8327703475952148, "sampling/importance_sampling_ratio/min": 0.15996979176998138, "sampling/importance_sampling_ratio/mean": 0.9933525919914246, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7010860815644264, "clip_ratio/low_mean": 0.055822014808654785, "clip_ratio/low_min": 0.055822014808654785, "clip_ratio/high_mean": 0.0606780918315053, "clip_ratio/high_max": 0.0606780918315053, "clip_ratio/region_mean": 0.11650010664016008, "reward_total_mean": 0.7391894459724426, "reward_meter_mean": 0.7711975574493408, "reward_meter_std": 0.2813066840171814, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9876835346221924, "reward_repeat_soft_std": 0.009118892252445221, "reward_judge_quality_mean": 0.7737500071525574, "reward_judge_quality_std": 0.22032040357589722, "reward_total_composite_mean": 0.7391894459724426, "reward_total_composite_std": 0.189705029129982} {"timestamp_utc": "2026-04-13T08:00:25Z", "mode": "train", "global_step": 269, "epoch": 0.027021597187343045, "loss": 0.0359, "grad_norm": 19.865617752075195, "learning_rate": 9.187878787878789e-06, "num_tokens": 498302.0, "completions/mean_length": 25.0, "completions/min_length": 17.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.0, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.903567910194397, "rewards/meter/std": 0.24369066953659058, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9612975120544434, "rewards/repeat_soft/std": 0.0034009767696261406, "rewards/judge_quality/mean": 0.48374998569488525, "rewards/judge_quality/std": 0.18965664505958557, "rewards/total_composite/mean": 0.6303813457489014, "rewards/total_composite/std": 0.14658081531524658, "reward": 0.6303813457489014, "reward_std": 0.1465807855129242, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20733590424060822, "sampling/sampling_logp_difference/max": 1.2930784225463867, "sampling/importance_sampling_ratio/min": 0.27442467212677, "sampling/importance_sampling_ratio/mean": 1.0204358100891113, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.115874543786049, "clip_ratio/low_mean": 0.023291926365345716, "clip_ratio/low_min": 0.023291926365345716, "clip_ratio/high_mean": 0.1650573741644621, "clip_ratio/high_max": 0.1650573741644621, "clip_ratio/region_mean": 0.1883493005298078, "reward_total_mean": 0.6303813457489014, "reward_meter_mean": 0.903567910194397, "reward_meter_std": 0.24369066953659058, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9612975120544434, "reward_repeat_soft_std": 0.0034009767696261406, "reward_judge_quality_mean": 0.48374998569488525, "reward_judge_quality_std": 0.18965664505958557, "reward_total_composite_mean": 0.6303813457489014, "reward_total_composite_std": 0.14658081531524658} {"timestamp_utc": "2026-04-13T08:00:32Z", "mode": "train", "global_step": 270, "epoch": 0.027122049221496736, "loss": -0.0944, "grad_norm": 13.092336654663086, "learning_rate": 9.184848484848485e-06, "num_tokens": 499890.0, "completions/mean_length": 43.5, "completions/min_length": 27.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.5, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.45151951909065247, "rewards/meter/std": 0.3828032910823822, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 1.0, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.38874998688697815, "rewards/judge_quality/std": 0.08675704896450043, "rewards/total_composite/mean": 0.4131723642349243, "rewards/total_composite/std": 0.18584245443344116, "reward": 0.4131723642349243, "reward_std": 0.18584245443344116, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2269778698682785, "sampling/sampling_logp_difference/max": 1.8083457946777344, "sampling/importance_sampling_ratio/min": 0.16392506659030914, "sampling/importance_sampling_ratio/mean": 1.041564702987671, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.1637730598449707, "clip_ratio/low_mean": 0.07881944440305233, "clip_ratio/low_min": 0.07881944440305233, "clip_ratio/high_mean": 0.13781122490763664, "clip_ratio/high_max": 0.13781122490763664, "clip_ratio/region_mean": 0.21663066931068897, "reward_total_mean": 0.4131723642349243, "reward_meter_mean": 0.45151951909065247, "reward_meter_std": 0.3828032910823822, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 1.0, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.38874998688697815, "reward_judge_quality_std": 0.08675704896450043, "reward_total_composite_mean": 0.4131723642349243, "reward_total_composite_std": 0.18584245443344116} {"timestamp_utc": "2026-04-13T08:00:39Z", "mode": "train", "global_step": 271, "epoch": 0.027222501255650428, "loss": 0.0151, "grad_norm": 14.901123046875, "learning_rate": 9.181818181818184e-06, "num_tokens": 501475.0, "completions/mean_length": 38.125, "completions/min_length": 31.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.125, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.5740398168563843, "rewards/meter/std": 0.3888934552669525, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9911396503448486, "rewards/repeat_soft/std": 0.006866309326142073, "rewards/judge_quality/mean": 0.3712500035762787, "rewards/judge_quality/std": 0.10091544687747955, "rewards/total_composite/mean": 0.5012853741645813, "rewards/total_composite/std": 0.11350028216838837, "reward": 0.5012853741645813, "reward_std": 0.11350028216838837, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2077735960483551, "sampling/sampling_logp_difference/max": 1.6415634155273438, "sampling/importance_sampling_ratio/min": 0.19367700815200806, "sampling/importance_sampling_ratio/mean": 1.0471086502075195, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.64426988363266, "clip_ratio/low_mean": 0.07613284746184945, "clip_ratio/low_min": 0.07613284746184945, "clip_ratio/high_mean": 0.09764429181814194, "clip_ratio/high_max": 0.09764429181814194, "clip_ratio/region_mean": 0.1737771392799914, "reward_total_mean": 0.5012853741645813, "reward_meter_mean": 0.5740398168563843, "reward_meter_std": 0.3888934552669525, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9911396503448486, "reward_repeat_soft_std": 0.006866309326142073, "reward_judge_quality_mean": 0.3712500035762787, "reward_judge_quality_std": 0.10091544687747955, "reward_total_composite_mean": 0.5012853741645813, "reward_total_composite_std": 0.11350028216838837} {"timestamp_utc": "2026-04-13T08:00:46Z", "mode": "train", "global_step": 272, "epoch": 0.02732295328980412, "loss": 0.1787, "grad_norm": 12.871830940246582, "learning_rate": 9.178787878787879e-06, "num_tokens": 503081.0, "completions/mean_length": 43.75, "completions/min_length": 38.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.75, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9809567928314209, "rewards/meter/std": 0.020499568432569504, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9782898426055908, "rewards/repeat_soft/std": 0.02181963250041008, "rewards/judge_quality/mean": 0.4112499952316284, "rewards/judge_quality/std": 0.1860443502664566, "rewards/total_composite/mean": 0.609815239906311, "rewards/total_composite/std": 0.12109959870576859, "reward": 0.609815239906311, "reward_std": 0.12109959870576859, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.25148677825927734, "sampling/sampling_logp_difference/max": 1.4036140441894531, "sampling/importance_sampling_ratio/min": 0.24570736289024353, "sampling/importance_sampling_ratio/mean": 1.060373306274414, "sampling/importance_sampling_ratio/max": 1.7493022680282593, "entropy": 3.5093444883823395, "clip_ratio/low_mean": 0.05145998392254114, "clip_ratio/low_min": 0.05145998392254114, "clip_ratio/high_mean": 0.16260988265275955, "clip_ratio/high_max": 0.16260988265275955, "clip_ratio/region_mean": 0.2140698665753007, "reward_total_mean": 0.609815239906311, "reward_meter_mean": 0.9809567928314209, "reward_meter_std": 0.020499568432569504, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9782898426055908, "reward_repeat_soft_std": 0.02181963250041008, "reward_judge_quality_mean": 0.4112499952316284, "reward_judge_quality_std": 0.1860443502664566, "reward_total_composite_mean": 0.609815239906311, "reward_total_composite_std": 0.12109959870576859} {"timestamp_utc": "2026-04-13T08:00:53Z", "mode": "train", "global_step": 273, "epoch": 0.02742340532395781, "loss": -0.0372, "grad_norm": 11.124080657958984, "learning_rate": 9.175757575757576e-06, "num_tokens": 504972.0, "completions/mean_length": 70.375, "completions/min_length": 61.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.375, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.5856324434280396, "rewards/meter/std": 0.3497793972492218, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9933459758758545, "rewards/repeat_soft/std": 0.006139182019978762, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.515045702457428, "rewards/total_composite/std": 0.09783965349197388, "reward": 0.515045702457428, "reward_std": 0.09783965349197388, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1920226514339447, "sampling/sampling_logp_difference/max": 1.841231346130371, "sampling/importance_sampling_ratio/min": 0.15862199664115906, "sampling/importance_sampling_ratio/mean": 1.0478473901748657, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1079849153757095, "clip_ratio/low_mean": 0.0917767696082592, "clip_ratio/low_min": 0.0917767696082592, "clip_ratio/high_mean": 0.08189899660646915, "clip_ratio/high_max": 0.08189899660646915, "clip_ratio/region_mean": 0.17367576621472836, "reward_total_mean": 0.515045702457428, "reward_meter_mean": 0.5856324434280396, "reward_meter_std": 0.3497793972492218, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9933459758758545, "reward_repeat_soft_std": 0.006139182019978762, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.515045702457428, "reward_total_composite_std": 0.09783965349197388} {"timestamp_utc": "2026-04-13T08:00:59Z", "mode": "train", "global_step": 274, "epoch": 0.0275238573581115, "loss": 0.1249, "grad_norm": 18.43695068359375, "learning_rate": 9.172727272727274e-06, "num_tokens": 506379.0, "completions/mean_length": 22.875, "completions/min_length": 18.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.875, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.6266977190971375, "rewards/meter/std": 0.4282899796962738, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9495941400527954, "rewards/repeat_soft/std": 0.03650323674082756, "rewards/judge_quality/mean": 0.3812499940395355, "rewards/judge_quality/std": 0.08166787773370743, "rewards/total_composite/mean": 0.5095803737640381, "rewards/total_composite/std": 0.12357732653617859, "reward": 0.5095803737640381, "reward_std": 0.12357732653617859, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23109155893325806, "sampling/sampling_logp_difference/max": 1.5454392433166504, "sampling/importance_sampling_ratio/min": 0.21321819722652435, "sampling/importance_sampling_ratio/mean": 1.0291531085968018, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1662951186299324, "clip_ratio/low_mean": 0.10648148227483034, "clip_ratio/low_min": 0.10648148227483034, "clip_ratio/high_mean": 0.11796537041664124, "clip_ratio/high_max": 0.11796537041664124, "clip_ratio/region_mean": 0.22444685269147158, "reward_total_mean": 0.5095803737640381, "reward_meter_mean": 0.6266977190971375, "reward_meter_std": 0.4282899796962738, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9495941400527954, "reward_repeat_soft_std": 0.03650323674082756, "reward_judge_quality_mean": 0.3812499940395355, "reward_judge_quality_std": 0.08166787773370743, "reward_total_composite_mean": 0.5095803737640381, "reward_total_composite_std": 0.12357732653617859} {"timestamp_utc": "2026-04-13T08:01:08Z", "mode": "train", "global_step": 275, "epoch": 0.027624309392265192, "loss": 0.1211, "grad_norm": 10.326603889465332, "learning_rate": 9.169696969696971e-06, "num_tokens": 508478.0, "completions/mean_length": 91.375, "completions/min_length": 69.0, "completions/max_length": 151.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.375, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 151.0, "rewards/meter/mean": 0.4572254419326782, "rewards/meter/std": 0.34324416518211365, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9886662364006042, "rewards/repeat_soft/std": 0.013447579927742481, "rewards/judge_quality/mean": 0.36124998331069946, "rewards/judge_quality/std": 0.1141975075006485, "rewards/total_composite/mean": 0.3660357594490051, "rewards/total_composite/std": 0.24225392937660217, "reward": 0.3660357594490051, "reward_std": 0.24225391447544098, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21642006933689117, "sampling/sampling_logp_difference/max": 1.1390104293823242, "sampling/importance_sampling_ratio/min": 0.3201356828212738, "sampling/importance_sampling_ratio/mean": 1.062514066696167, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.1374784111976624, "clip_ratio/low_mean": 0.05686148162931204, "clip_ratio/low_min": 0.05686148162931204, "clip_ratio/high_mean": 0.12252083979547024, "clip_ratio/high_max": 0.12252083979547024, "clip_ratio/region_mean": 0.17938232142478228, "reward_total_mean": 0.3660357594490051, "reward_meter_mean": 0.4572254419326782, "reward_meter_std": 0.34324416518211365, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9886662364006042, "reward_repeat_soft_std": 0.013447579927742481, "reward_judge_quality_mean": 0.36124998331069946, "reward_judge_quality_std": 0.1141975075006485, "reward_total_composite_mean": 0.3660357594490051, "reward_total_composite_std": 0.24225392937660217} {"timestamp_utc": "2026-04-13T08:01:15Z", "mode": "train", "global_step": 276, "epoch": 0.027724761426418883, "loss": 0.3342, "grad_norm": 20.758947372436523, "learning_rate": 9.166666666666666e-06, "num_tokens": 510069.0, "completions/mean_length": 38.875, "completions/min_length": 30.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.875, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.7617350816726685, "rewards/meter/std": 0.314482718706131, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9918779134750366, "rewards/repeat_soft/std": 0.01899106800556183, "rewards/judge_quality/mean": 0.44749999046325684, "rewards/judge_quality/std": 0.12848013639450073, "rewards/total_composite/mean": 0.5732759237289429, "rewards/total_composite/std": 0.149704709649086, "reward": 0.5732759237289429, "reward_std": 0.1497046947479248, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.241552472114563, "sampling/sampling_logp_difference/max": 1.3904485702514648, "sampling/importance_sampling_ratio/min": 0.2489636093378067, "sampling/importance_sampling_ratio/mean": 1.0301902294158936, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.613171875476837, "clip_ratio/low_mean": 0.06453399732708931, "clip_ratio/low_min": 0.06453399732708931, "clip_ratio/high_mean": 0.13725490681827068, "clip_ratio/high_max": 0.13725490681827068, "clip_ratio/region_mean": 0.20178890414536, "reward_total_mean": 0.5732759237289429, "reward_meter_mean": 0.7617350816726685, "reward_meter_std": 0.314482718706131, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9918779134750366, "reward_repeat_soft_std": 0.01899106800556183, "reward_judge_quality_mean": 0.44749999046325684, "reward_judge_quality_std": 0.12848013639450073, "reward_total_composite_mean": 0.5732759237289429, "reward_total_composite_std": 0.149704709649086} {"timestamp_utc": "2026-04-13T08:01:22Z", "mode": "train", "global_step": 277, "epoch": 0.027825213460572578, "loss": -0.0043, "grad_norm": 19.438020706176758, "learning_rate": 9.163636363636365e-06, "num_tokens": 511499.0, "completions/mean_length": 32.75, "completions/min_length": 26.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.75, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.7640655040740967, "rewards/meter/std": 0.31606751680374146, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9978388547897339, "rewards/repeat_soft/std": 0.0030812383629381657, "rewards/judge_quality/mean": 0.7075000405311584, "rewards/judge_quality/std": 0.21001702547073364, "rewards/total_composite/mean": 0.7123679518699646, "rewards/total_composite/std": 0.16456754505634308, "reward": 0.7123679518699646, "reward_std": 0.16456754505634308, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20448407530784607, "sampling/sampling_logp_difference/max": 1.4917001724243164, "sampling/importance_sampling_ratio/min": 0.22498980164527893, "sampling/importance_sampling_ratio/mean": 1.0104999542236328, "sampling/importance_sampling_ratio/max": 1.8533438444137573, "entropy": 1.5110199749469757, "clip_ratio/low_mean": 0.035912297666072845, "clip_ratio/low_min": 0.035912297666072845, "clip_ratio/high_mean": 0.15360071696341038, "clip_ratio/high_max": 0.15360071696341038, "clip_ratio/region_mean": 0.18951301462948322, "reward_total_mean": 0.7123679518699646, "reward_meter_mean": 0.7640655040740967, "reward_meter_std": 0.31606751680374146, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9978388547897339, "reward_repeat_soft_std": 0.0030812383629381657, "reward_judge_quality_mean": 0.7075000405311584, "reward_judge_quality_std": 0.21001702547073364, "reward_total_composite_mean": 0.7123679518699646, "reward_total_composite_std": 0.16456754505634308} {"timestamp_utc": "2026-04-13T08:01:29Z", "mode": "train", "global_step": 278, "epoch": 0.02792566549472627, "loss": 0.0259, "grad_norm": 34.60313415527344, "learning_rate": 9.160606060606061e-06, "num_tokens": 512918.0, "completions/mean_length": 24.375, "completions/min_length": 21.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.375, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.9747086763381958, "rewards/meter/std": 0.05066248029470444, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9778183102607727, "rewards/repeat_soft/std": 0.021749334409832954, "rewards/judge_quality/mean": 0.9200000166893005, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.9295485019683838, "rewards/total_composite/std": 0.03065398521721363, "reward": 0.9295485019683838, "reward_std": 0.03065398521721363, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2493032068014145, "sampling/sampling_logp_difference/max": 2.6578667163848877, "sampling/importance_sampling_ratio/min": 0.07009760290384293, "sampling/importance_sampling_ratio/mean": 0.9814552068710327, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.998925194144249, "clip_ratio/low_mean": 0.02604166604578495, "clip_ratio/low_min": 0.02604166604578495, "clip_ratio/high_mean": 0.16956652328372002, "clip_ratio/high_max": 0.16956652328372002, "clip_ratio/region_mean": 0.19560818932950497, "reward_total_mean": 0.9295485019683838, "reward_meter_mean": 0.9747086763381958, "reward_meter_std": 0.05066248029470444, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9778183102607727, "reward_repeat_soft_std": 0.021749334409832954, "reward_judge_quality_mean": 0.9200000166893005, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.9295485019683838, "reward_total_composite_std": 0.03065398521721363} {"timestamp_utc": "2026-04-13T08:01:37Z", "mode": "train", "global_step": 279, "epoch": 0.02802611752887996, "loss": 0.069, "grad_norm": 12.388896942138672, "learning_rate": 9.157575757575758e-06, "num_tokens": 514522.0, "completions/mean_length": 45.5, "completions/min_length": 34.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.5, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.8167272806167603, "rewards/meter/std": 0.3358098864555359, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9955008029937744, "rewards/repeat_soft/std": 0.006966626737266779, "rewards/judge_quality/mean": 0.5062500238418579, "rewards/judge_quality/std": 0.25999659299850464, "rewards/total_composite/mean": 0.6498563885688782, "rewards/total_composite/std": 0.20248964428901672, "reward": 0.6498563885688782, "reward_std": 0.20248964428901672, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20259158313274384, "sampling/sampling_logp_difference/max": 1.0752041339874268, "sampling/importance_sampling_ratio/min": 0.36106279492378235, "sampling/importance_sampling_ratio/mean": 1.0507522821426392, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.3180495500564575, "clip_ratio/low_mean": 0.1202126294374466, "clip_ratio/low_min": 0.1202126294374466, "clip_ratio/high_mean": 0.08005169313400984, "clip_ratio/high_max": 0.08005169313400984, "clip_ratio/region_mean": 0.20026432257145643, "reward_total_mean": 0.6498563885688782, "reward_meter_mean": 0.8167272806167603, "reward_meter_std": 0.3358098864555359, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9955008029937744, "reward_repeat_soft_std": 0.006966626737266779, "reward_judge_quality_mean": 0.5062500238418579, "reward_judge_quality_std": 0.25999659299850464, "reward_total_composite_mean": 0.6498563885688782, "reward_total_composite_std": 0.20248964428901672} {"timestamp_utc": "2026-04-13T08:01:45Z", "mode": "train", "global_step": 280, "epoch": 0.02812656956303365, "loss": 0.102, "grad_norm": 7.115893363952637, "learning_rate": 9.154545454545455e-06, "num_tokens": 516764.0, "completions/mean_length": 114.25, "completions/min_length": 93.0, "completions/max_length": 150.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 114.25, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 150.0, "rewards/meter/mean": 0.8475615978240967, "rewards/meter/std": 0.24203528463840485, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9811174869537354, "rewards/repeat_soft/std": 0.020043261349201202, "rewards/judge_quality/mean": 0.26874998211860657, "rewards/judge_quality/std": 0.13505950570106506, "rewards/total_composite/mean": 0.46794646978378296, "rewards/total_composite/std": 0.09037885069847107, "reward": 0.46794646978378296, "reward_std": 0.09037884324789047, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23452864587306976, "sampling/sampling_logp_difference/max": 1.3950977325439453, "sampling/importance_sampling_ratio/min": 0.24780882894992828, "sampling/importance_sampling_ratio/mean": 1.0599229335784912, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.9277411103248596, "clip_ratio/low_mean": 0.07516323309391737, "clip_ratio/low_min": 0.07516323309391737, "clip_ratio/high_mean": 0.07142539694905281, "clip_ratio/high_max": 0.07142539694905281, "clip_ratio/region_mean": 0.14658863004297018, "reward_total_mean": 0.46794646978378296, "reward_meter_mean": 0.8475615978240967, "reward_meter_std": 0.24203528463840485, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9811174869537354, "reward_repeat_soft_std": 0.020043261349201202, "reward_judge_quality_mean": 0.26874998211860657, "reward_judge_quality_std": 0.13505950570106506, "reward_total_composite_mean": 0.46794646978378296, "reward_total_composite_std": 0.09037885069847107} {"timestamp_utc": "2026-04-13T08:01:52Z", "mode": "train", "global_step": 281, "epoch": 0.028227021597187343, "loss": 0.0678, "grad_norm": 13.96617317199707, "learning_rate": 9.151515151515153e-06, "num_tokens": 518319.0, "completions/mean_length": 40.375, "completions/min_length": 37.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.375, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.9172787666320801, "rewards/meter/std": 0.10134030878543854, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.989300549030304, "rewards/repeat_soft/std": 0.013068420812487602, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.6098494529724121, "rewards/total_composite/std": 0.02921118587255478, "reward": 0.6098494529724121, "reward_std": 0.02921118400990963, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20314764976501465, "sampling/sampling_logp_difference/max": 1.6208076477050781, "sampling/importance_sampling_ratio/min": 0.19773893058300018, "sampling/importance_sampling_ratio/mean": 1.0205670595169067, "sampling/importance_sampling_ratio/max": 1.9314621686935425, "entropy": 2.173624560236931, "clip_ratio/low_mean": 0.032751716673374176, "clip_ratio/low_min": 0.032751716673374176, "clip_ratio/high_mean": 0.14773373864591122, "clip_ratio/high_max": 0.14773373864591122, "clip_ratio/region_mean": 0.1804854553192854, "reward_total_mean": 0.6098494529724121, "reward_meter_mean": 0.9172787666320801, "reward_meter_std": 0.10134030878543854, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.989300549030304, "reward_repeat_soft_std": 0.013068420812487602, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.6098494529724121, "reward_total_composite_std": 0.02921118587255478} {"timestamp_utc": "2026-04-13T08:01:59Z", "mode": "train", "global_step": 282, "epoch": 0.028327473631341034, "loss": 0.2188, "grad_norm": 14.918357849121094, "learning_rate": 9.148484848484848e-06, "num_tokens": 519838.0, "completions/mean_length": 39.875, "completions/min_length": 29.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.875, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.35328733921051025, "rewards/meter/std": 0.3507688641548157, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9957976937294006, "rewards/repeat_soft/std": 0.009099088609218597, "rewards/judge_quality/mean": 0.5099999904632568, "rewards/judge_quality/std": 0.215406596660614, "rewards/total_composite/mean": 0.38957157731056213, "rewards/total_composite/std": 0.18033328652381897, "reward": 0.38957157731056213, "reward_std": 0.18033325672149658, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23948487639427185, "sampling/sampling_logp_difference/max": 1.5781440734863281, "sampling/importance_sampling_ratio/min": 0.20635771751403809, "sampling/importance_sampling_ratio/mean": 1.0777320861816406, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4532696157693863, "clip_ratio/low_mean": 0.09098005574196577, "clip_ratio/low_min": 0.09098005574196577, "clip_ratio/high_mean": 0.09778170101344585, "clip_ratio/high_max": 0.09778170101344585, "clip_ratio/region_mean": 0.18876175675541162, "reward_total_mean": 0.38957157731056213, "reward_meter_mean": 0.35328733921051025, "reward_meter_std": 0.3507688641548157, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9957976937294006, "reward_repeat_soft_std": 0.009099088609218597, "reward_judge_quality_mean": 0.5099999904632568, "reward_judge_quality_std": 0.215406596660614, "reward_total_composite_mean": 0.38957157731056213, "reward_total_composite_std": 0.18033328652381897} {"timestamp_utc": "2026-04-13T08:02:07Z", "mode": "train", "global_step": 283, "epoch": 0.028427925665494725, "loss": 0.099, "grad_norm": 9.182372093200684, "learning_rate": 9.145454545454546e-06, "num_tokens": 522270.0, "completions/mean_length": 124.0, "completions/min_length": 74.0, "completions/max_length": 164.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 124.0, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 164.0, "rewards/meter/mean": 0.5346735715866089, "rewards/meter/std": 0.31673410534858704, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.07715168595314026, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9935542345046997, "rewards/repeat_soft/std": 0.008262968622148037, "rewards/judge_quality/mean": 0.26374998688697815, "rewards/judge_quality/std": 0.10901343077421188, "rewards/total_composite/mean": 0.37938377261161804, "rewards/total_composite/std": 0.17373378574848175, "reward": 0.37938377261161804, "reward_std": 0.17373378574848175, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23359964787960052, "sampling/sampling_logp_difference/max": 1.3233938217163086, "sampling/importance_sampling_ratio/min": 0.2662302553653717, "sampling/importance_sampling_ratio/mean": 1.052581548690796, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.300060272216797, "clip_ratio/low_mean": 0.09318131022155285, "clip_ratio/low_min": 0.09318131022155285, "clip_ratio/high_mean": 0.1137061920017004, "clip_ratio/high_max": 0.1137061920017004, "clip_ratio/region_mean": 0.20688750222325325, "reward_total_mean": 0.37938377261161804, "reward_meter_mean": 0.5346735715866089, "reward_meter_std": 0.31673410534858704, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.07715168595314026, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9935542345046997, "reward_repeat_soft_std": 0.008262968622148037, "reward_judge_quality_mean": 0.26374998688697815, "reward_judge_quality_std": 0.10901343077421188, "reward_total_composite_mean": 0.37938377261161804, "reward_total_composite_std": 0.17373378574848175} {"timestamp_utc": "2026-04-13T08:02:15Z", "mode": "train", "global_step": 284, "epoch": 0.028528377699648416, "loss": 0.1768, "grad_norm": 10.7915620803833, "learning_rate": 9.142424242424243e-06, "num_tokens": 524321.0, "completions/mean_length": 81.375, "completions/min_length": 56.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 81.375, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.3346508741378784, "rewards/meter/std": 0.24749335646629333, "rewards/count_adherence/mean": 0.84375, "rewards/count_adherence/std": 0.18600596487522125, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9945473670959473, "rewards/repeat_soft/std": 0.004762158263474703, "rewards/judge_quality/mean": 0.4737499952316284, "rewards/judge_quality/std": 0.16291432082653046, "rewards/total_composite/mean": 0.426018089056015, "rewards/total_composite/std": 0.09572703391313553, "reward": 0.426018089056015, "reward_std": 0.09572703391313553, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23324817419052124, "sampling/sampling_logp_difference/max": 1.6429443359375, "sampling/importance_sampling_ratio/min": 0.193409726023674, "sampling/importance_sampling_ratio/mean": 1.0439000129699707, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6393192410469055, "clip_ratio/low_mean": 0.08876687660813332, "clip_ratio/low_min": 0.08876687660813332, "clip_ratio/high_mean": 0.11488107591867447, "clip_ratio/high_max": 0.11488107591867447, "clip_ratio/region_mean": 0.20364795252680779, "reward_total_mean": 0.426018089056015, "reward_meter_mean": 0.3346508741378784, "reward_meter_std": 0.24749335646629333, "reward_count_adherence_mean": 0.84375, "reward_count_adherence_std": 0.18600596487522125, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9945473670959473, "reward_repeat_soft_std": 0.004762158263474703, "reward_judge_quality_mean": 0.4737499952316284, "reward_judge_quality_std": 0.16291432082653046, "reward_total_composite_mean": 0.426018089056015, "reward_total_composite_std": 0.09572703391313553} {"timestamp_utc": "2026-04-13T08:02:22Z", "mode": "train", "global_step": 285, "epoch": 0.02862882973380211, "loss": 0.0998, "grad_norm": 18.474149703979492, "learning_rate": 9.13939393939394e-06, "num_tokens": 525809.0, "completions/mean_length": 24.0, "completions/min_length": 19.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.0, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.5993695855140686, "rewards/meter/std": 0.4427065849304199, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9621384143829346, "rewards/repeat_soft/std": 0.0010226722806692123, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.0645727664232254, "rewards/total_composite/mean": 0.5071156620979309, "rewards/total_composite/std": 0.12619905173778534, "reward": 0.5071156620979309, "reward_std": 0.12619903683662415, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19200460612773895, "sampling/sampling_logp_difference/max": 1.185643196105957, "sampling/importance_sampling_ratio/min": 0.30554959177970886, "sampling/importance_sampling_ratio/mean": 1.0759772062301636, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0072427690029144, "clip_ratio/low_mean": 0.07683908101171255, "clip_ratio/low_min": 0.07683908101171255, "clip_ratio/high_mean": 0.08359994180500507, "clip_ratio/high_max": 0.08359994180500507, "clip_ratio/region_mean": 0.16043902281671762, "reward_total_mean": 0.5071156620979309, "reward_meter_mean": 0.5993695855140686, "reward_meter_std": 0.4427065849304199, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9621384143829346, "reward_repeat_soft_std": 0.0010226722806692123, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.0645727664232254, "reward_total_composite_mean": 0.5071156620979309, "reward_total_composite_std": 0.12619905173778534} {"timestamp_utc": "2026-04-13T08:02:34Z", "mode": "train", "global_step": 286, "epoch": 0.028729281767955802, "loss": -0.1503, "grad_norm": 2.696510076522827, "learning_rate": 9.136363636363637e-06, "num_tokens": 527855.0, "completions/mean_length": 191.75, "completions/min_length": 67.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 85.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.22511011362075806, "rewards/meter/std": 0.22926172614097595, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9987255334854126, "rewards/repeat_soft/std": 0.001067682751454413, "rewards/judge_quality/mean": 0.2224999964237213, "rewards/judge_quality/std": 0.16799233853816986, "rewards/total_composite/mean": 0.283405065536499, "rewards/total_composite/std": 0.19441770017147064, "reward": 0.283405065536499, "reward_std": 0.19441771507263184, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24778270721435547, "sampling/sampling_logp_difference/max": 1.1723203659057617, "sampling/importance_sampling_ratio/min": 0.3096476197242737, "sampling/importance_sampling_ratio/mean": 1.1020667552947998, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.5178698301315308, "clip_ratio/low_mean": 0.023989899083971977, "clip_ratio/low_min": 0.023989899083971977, "clip_ratio/high_mean": 0.11538269557058811, "clip_ratio/high_max": 0.11538269557058811, "clip_ratio/region_mean": 0.1393725946545601, "reward_total_mean": 0.283405065536499, "reward_meter_mean": 0.22511011362075806, "reward_meter_std": 0.22926172614097595, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9987255334854126, "reward_repeat_soft_std": 0.001067682751454413, "reward_judge_quality_mean": 0.2224999964237213, "reward_judge_quality_std": 0.16799233853816986, "reward_total_composite_mean": 0.283405065536499, "reward_total_composite_std": 0.19441770017147064} {"timestamp_utc": "2026-04-13T08:02:46Z", "mode": "train", "global_step": 287, "epoch": 0.028829733802109493, "loss": -0.191, "grad_norm": 2.264235496520996, "learning_rate": 9.133333333333335e-06, "num_tokens": 530125.0, "completions/mean_length": 237.75, "completions/min_length": 121.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 146.33334350585938, "completions/min_terminated_length": 121.0, "completions/max_terminated_length": 181.0, "rewards/meter/mean": 0.4126812219619751, "rewards/meter/std": 0.27017462253570557, "rewards/count_adherence/mean": 0.7750000357627869, "rewards/count_adherence/std": 0.16690459847450256, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.999116063117981, "rewards/repeat_soft/std": 0.0009569707908667624, "rewards/judge_quality/mean": 0.2512499988079071, "rewards/judge_quality/std": 0.15887439250946045, "rewards/total_composite/mean": 0.33408790826797485, "rewards/total_composite/std": 0.16096235811710358, "reward": 0.33408790826797485, "reward_std": 0.16096235811710358, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23185819387435913, "sampling/sampling_logp_difference/max": 1.263564109802246, "sampling/importance_sampling_ratio/min": 0.2826448380947113, "sampling/importance_sampling_ratio/mean": 1.054463505744934, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.1233180463314056, "clip_ratio/low_mean": 0.006369426846504211, "clip_ratio/low_min": 0.006369426846504211, "clip_ratio/high_mean": 0.13328539207577705, "clip_ratio/high_max": 0.13328539207577705, "clip_ratio/region_mean": 0.13965481892228127, "reward_total_mean": 0.33408790826797485, "reward_meter_mean": 0.4126812219619751, "reward_meter_std": 0.27017462253570557, "reward_count_adherence_mean": 0.7750000357627869, "reward_count_adherence_std": 0.16690459847450256, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.999116063117981, "reward_repeat_soft_std": 0.0009569707908667624, "reward_judge_quality_mean": 0.2512499988079071, "reward_judge_quality_std": 0.15887439250946045, "reward_total_composite_mean": 0.33408790826797485, "reward_total_composite_std": 0.16096235811710358} {"timestamp_utc": "2026-04-13T08:02:53Z", "mode": "train", "global_step": 288, "epoch": 0.028930185836263184, "loss": 0.0965, "grad_norm": 18.33196258544922, "learning_rate": 9.130303030303032e-06, "num_tokens": 531671.0, "completions/mean_length": 38.25, "completions/min_length": 31.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.25, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.5347582101821899, "rewards/meter/std": 0.4431576132774353, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9937200546264648, "rewards/repeat_soft/std": 0.005315437447279692, "rewards/judge_quality/mean": 0.5225000381469727, "rewards/judge_quality/std": 0.2406984120607376, "rewards/total_composite/mean": 0.5168812870979309, "rewards/total_composite/std": 0.26617226004600525, "reward": 0.5168812870979309, "reward_std": 0.26617226004600525, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22125428915023804, "sampling/sampling_logp_difference/max": 1.2057805061340332, "sampling/importance_sampling_ratio/min": 0.29945817589759827, "sampling/importance_sampling_ratio/mean": 1.0432169437408447, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.7232500463724136, "clip_ratio/low_mean": 0.08229638077318668, "clip_ratio/low_min": 0.08229638077318668, "clip_ratio/high_mean": 0.11801720969378948, "clip_ratio/high_max": 0.11801720969378948, "clip_ratio/region_mean": 0.20031359046697617, "reward_total_mean": 0.5168812870979309, "reward_meter_mean": 0.5347582101821899, "reward_meter_std": 0.4431576132774353, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9937200546264648, "reward_repeat_soft_std": 0.005315437447279692, "reward_judge_quality_mean": 0.5225000381469727, "reward_judge_quality_std": 0.2406984120607376, "reward_total_composite_mean": 0.5168812870979309, "reward_total_composite_std": 0.26617226004600525} {"timestamp_utc": "2026-04-13T08:03:01Z", "mode": "train", "global_step": 289, "epoch": 0.029030637870416875, "loss": 0.0646, "grad_norm": 16.360633850097656, "learning_rate": 9.127272727272727e-06, "num_tokens": 533543.0, "completions/mean_length": 70.0, "completions/min_length": 45.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.0, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.5416287183761597, "rewards/meter/std": 0.33586210012435913, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9970144033432007, "rewards/repeat_soft/std": 0.0027460602577775717, "rewards/judge_quality/mean": 0.6200000047683716, "rewards/judge_quality/std": 0.21380899846553802, "rewards/total_composite/mean": 0.5797946453094482, "rewards/total_composite/std": 0.17210178077220917, "reward": 0.5797946453094482, "reward_std": 0.17210179567337036, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2408500611782074, "sampling/sampling_logp_difference/max": 1.9898395538330078, "sampling/importance_sampling_ratio/min": 0.13671736419200897, "sampling/importance_sampling_ratio/mean": 1.0045815706253052, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4601524621248245, "clip_ratio/low_mean": 0.09364362061023712, "clip_ratio/low_min": 0.09364362061023712, "clip_ratio/high_mean": 0.10282890498638153, "clip_ratio/high_max": 0.10282890498638153, "clip_ratio/region_mean": 0.19647252559661865, "reward_total_mean": 0.5797946453094482, "reward_meter_mean": 0.5416287183761597, "reward_meter_std": 0.33586210012435913, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9970144033432007, "reward_repeat_soft_std": 0.0027460602577775717, "reward_judge_quality_mean": 0.6200000047683716, "reward_judge_quality_std": 0.21380899846553802, "reward_total_composite_mean": 0.5797946453094482, "reward_total_composite_std": 0.17210178077220917} {"timestamp_utc": "2026-04-13T08:03:08Z", "mode": "train", "global_step": 290, "epoch": 0.029131089904570567, "loss": 0.0531, "grad_norm": 13.182306289672852, "learning_rate": 9.124242424242425e-06, "num_tokens": 535226.0, "completions/mean_length": 44.375, "completions/min_length": 40.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.375, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.9862470626831055, "rewards/meter/std": 0.010380273684859276, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9979305863380432, "rewards/repeat_soft/std": 0.003956858534365892, "rewards/judge_quality/mean": 0.6775000095367432, "rewards/judge_quality/std": 0.26762181520462036, "rewards/total_composite/mean": 0.7850335836410522, "rewards/total_composite/std": 0.17486260831356049, "reward": 0.7850335836410522, "reward_std": 0.17486260831356049, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1653902381658554, "sampling/sampling_logp_difference/max": 1.4844894409179688, "sampling/importance_sampling_ratio/min": 0.22661800682544708, "sampling/importance_sampling_ratio/mean": 1.0317223072052002, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5290031135082245, "clip_ratio/low_mean": 0.06294930493459105, "clip_ratio/low_min": 0.06294930493459105, "clip_ratio/high_mean": 0.06362293194979429, "clip_ratio/high_max": 0.06362293194979429, "clip_ratio/region_mean": 0.12657223688438535, "reward_total_mean": 0.7850335836410522, "reward_meter_mean": 0.9862470626831055, "reward_meter_std": 0.010380273684859276, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9979305863380432, "reward_repeat_soft_std": 0.003956858534365892, "reward_judge_quality_mean": 0.6775000095367432, "reward_judge_quality_std": 0.26762181520462036, "reward_total_composite_mean": 0.7850335836410522, "reward_total_composite_std": 0.17486260831356049} {"timestamp_utc": "2026-04-13T08:03:15Z", "mode": "train", "global_step": 291, "epoch": 0.029231541938724258, "loss": -0.0157, "grad_norm": 13.291495323181152, "learning_rate": 9.121212121212122e-06, "num_tokens": 536954.0, "completions/mean_length": 39.0, "completions/min_length": 30.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.0, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.7474240064620972, "rewards/meter/std": 0.3927259147167206, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9828974008560181, "rewards/repeat_soft/std": 0.016468608751893044, "rewards/judge_quality/mean": 0.3725000023841858, "rewards/judge_quality/std": 0.11055056750774384, "rewards/total_composite/mean": 0.5376838445663452, "rewards/total_composite/std": 0.11802530288696289, "reward": 0.5376838445663452, "reward_std": 0.1180252879858017, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20696787536144257, "sampling/sampling_logp_difference/max": 0.9229650497436523, "sampling/importance_sampling_ratio/min": 0.4048669636249542, "sampling/importance_sampling_ratio/mean": 1.0405880212783813, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.7855406999588013, "clip_ratio/low_mean": 0.09184157848358154, "clip_ratio/low_min": 0.09184157848358154, "clip_ratio/high_mean": 0.0927211381494999, "clip_ratio/high_max": 0.0927211381494999, "clip_ratio/region_mean": 0.18456271663308144, "reward_total_mean": 0.5376838445663452, "reward_meter_mean": 0.7474240064620972, "reward_meter_std": 0.3927259147167206, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9828974008560181, "reward_repeat_soft_std": 0.016468608751893044, "reward_judge_quality_mean": 0.3725000023841858, "reward_judge_quality_std": 0.11055056750774384, "reward_total_composite_mean": 0.5376838445663452, "reward_total_composite_std": 0.11802530288696289} {"timestamp_utc": "2026-04-13T08:03:23Z", "mode": "train", "global_step": 292, "epoch": 0.029331993972877952, "loss": 0.0883, "grad_norm": 10.179062843322754, "learning_rate": 9.118181818181819e-06, "num_tokens": 539060.0, "completions/mean_length": 82.25, "completions/min_length": 56.0, "completions/max_length": 112.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.25, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 112.0, "rewards/meter/mean": 0.4219987988471985, "rewards/meter/std": 0.2983505427837372, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.99605393409729, "rewards/repeat_soft/std": 0.002786876168102026, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.21357084810733795, "rewards/total_composite/mean": 0.4893326759338379, "rewards/total_composite/std": 0.10986044257879257, "reward": 0.4893326759338379, "reward_std": 0.10986043512821198, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23603616654872894, "sampling/sampling_logp_difference/max": 2.265481948852539, "sampling/importance_sampling_ratio/min": 0.10378000885248184, "sampling/importance_sampling_ratio/mean": 1.0513489246368408, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.693507432937622, "clip_ratio/low_mean": 0.1513797529041767, "clip_ratio/low_min": 0.1513797529041767, "clip_ratio/high_mean": 0.06801920384168625, "clip_ratio/high_max": 0.06801920384168625, "clip_ratio/region_mean": 0.21939895674586296, "reward_total_mean": 0.4893326759338379, "reward_meter_mean": 0.4219987988471985, "reward_meter_std": 0.2983505427837372, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.99605393409729, "reward_repeat_soft_std": 0.002786876168102026, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.21357084810733795, "reward_total_composite_mean": 0.4893326759338379, "reward_total_composite_std": 0.10986044257879257} {"timestamp_utc": "2026-04-13T08:03:30Z", "mode": "train", "global_step": 293, "epoch": 0.029432446007031644, "loss": -0.0211, "grad_norm": 15.809340476989746, "learning_rate": 9.115151515151516e-06, "num_tokens": 540735.0, "completions/mean_length": 34.375, "completions/min_length": 32.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.375, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.5477538108825684, "rewards/meter/std": 0.3368853032588959, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.996740460395813, "rewards/repeat_soft/std": 0.005701808724552393, "rewards/judge_quality/mean": 0.48624998331069946, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.5014572739601135, "rewards/total_composite/std": 0.09065254777669907, "reward": 0.5014572739601135, "reward_std": 0.09065254032611847, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20220136642456055, "sampling/sampling_logp_difference/max": 1.4310646057128906, "sampling/importance_sampling_ratio/min": 0.23905429244041443, "sampling/importance_sampling_ratio/mean": 1.0401586294174194, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.230154663324356, "clip_ratio/low_mean": 0.0651041679084301, "clip_ratio/low_min": 0.0651041679084301, "clip_ratio/high_mean": 0.12616414576768875, "clip_ratio/high_max": 0.12616414576768875, "clip_ratio/region_mean": 0.19126831367611885, "reward_total_mean": 0.5014572739601135, "reward_meter_mean": 0.5477538108825684, "reward_meter_std": 0.3368853032588959, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.996740460395813, "reward_repeat_soft_std": 0.005701808724552393, "reward_judge_quality_mean": 0.48624998331069946, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.5014572739601135, "reward_total_composite_std": 0.09065254777669907} {"timestamp_utc": "2026-04-13T08:03:40Z", "mode": "train", "global_step": 294, "epoch": 0.029532898041185335, "loss": 0.063, "grad_norm": 11.58062744140625, "learning_rate": 9.112121212121214e-06, "num_tokens": 542607.0, "completions/mean_length": 56.0, "completions/min_length": 45.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.0, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.6475329399108887, "rewards/meter/std": 0.3424645960330963, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9984492063522339, "rewards/repeat_soft/std": 0.0020309877581894398, "rewards/judge_quality/mean": 0.4049999713897705, "rewards/judge_quality/std": 0.16716118156909943, "rewards/total_composite/mean": 0.5189603567123413, "rewards/total_composite/std": 0.1285088062286377, "reward": 0.5189603567123413, "reward_std": 0.1285088211297989, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22326365113258362, "sampling/sampling_logp_difference/max": 1.262068748474121, "sampling/importance_sampling_ratio/min": 0.28306782245635986, "sampling/importance_sampling_ratio/mean": 1.0421050786972046, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6189921498298645, "clip_ratio/low_mean": 0.12590462155640125, "clip_ratio/low_min": 0.12590462155640125, "clip_ratio/high_mean": 0.07572508230805397, "clip_ratio/high_max": 0.07572508230805397, "clip_ratio/region_mean": 0.20162970386445522, "reward_total_mean": 0.5189603567123413, "reward_meter_mean": 0.6475329399108887, "reward_meter_std": 0.3424645960330963, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9984492063522339, "reward_repeat_soft_std": 0.0020309877581894398, "reward_judge_quality_mean": 0.4049999713897705, "reward_judge_quality_std": 0.16716118156909943, "reward_total_composite_mean": 0.5189603567123413, "reward_total_composite_std": 0.1285088062286377} {"timestamp_utc": "2026-04-13T08:03:46Z", "mode": "train", "global_step": 295, "epoch": 0.029633350075339026, "loss": 0.0501, "grad_norm": 14.44871997833252, "learning_rate": 9.10909090909091e-06, "num_tokens": 543998.0, "completions/mean_length": 19.875, "completions/min_length": 15.0, "completions/max_length": 24.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 19.875, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 24.0, "rewards/meter/mean": 0.7429407238960266, "rewards/meter/std": 0.37553855776786804, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9585226774215698, "rewards/repeat_soft/std": 0.011249415576457977, "rewards/judge_quality/mean": 0.4937500059604645, "rewards/judge_quality/std": 0.1728696972131729, "rewards/total_composite/mean": 0.5907266139984131, "rewards/total_composite/std": 0.1698986142873764, "reward": 0.5907266139984131, "reward_std": 0.1698986291885376, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18262533843517303, "sampling/sampling_logp_difference/max": 1.3003034591674805, "sampling/importance_sampling_ratio/min": 0.2724491059780121, "sampling/importance_sampling_ratio/mean": 1.0100187063217163, "sampling/importance_sampling_ratio/max": 1.863359808921814, "entropy": 1.568077489733696, "clip_ratio/low_mean": 0.053125000558793545, "clip_ratio/low_min": 0.053125000558793545, "clip_ratio/high_mean": 0.14995477348566055, "clip_ratio/high_max": 0.14995477348566055, "clip_ratio/region_mean": 0.2030797740444541, "reward_total_mean": 0.5907266139984131, "reward_meter_mean": 0.7429407238960266, "reward_meter_std": 0.37553855776786804, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9585226774215698, "reward_repeat_soft_std": 0.011249415576457977, "reward_judge_quality_mean": 0.4937500059604645, "reward_judge_quality_std": 0.1728696972131729, "reward_total_composite_mean": 0.5907266139984131, "reward_total_composite_std": 0.1698986142873764} {"timestamp_utc": "2026-04-13T08:03:53Z", "mode": "train", "global_step": 296, "epoch": 0.029733802109492717, "loss": 0.1309, "grad_norm": 22.907466888427734, "learning_rate": 9.106060606060606e-06, "num_tokens": 545273.0, "completions/mean_length": 19.375, "completions/min_length": 14.0, "completions/max_length": 25.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 19.375, "completions/min_terminated_length": 14.0, "completions/max_terminated_length": 25.0, "rewards/meter/mean": 0.4516580104827881, "rewards/meter/std": 0.4649323523044586, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8865073919296265, "rewards/repeat_soft/std": 0.18254618346691132, "rewards/judge_quality/mean": 0.38749998807907104, "rewards/judge_quality/std": 0.11877349019050598, "rewards/total_composite/mean": 0.4390692710876465, "rewards/total_composite/std": 0.12529818713665009, "reward": 0.4390692710876465, "reward_std": 0.12529820203781128, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21530936658382416, "sampling/sampling_logp_difference/max": 1.4255495071411133, "sampling/importance_sampling_ratio/min": 0.2403763383626938, "sampling/importance_sampling_ratio/mean": 1.0328723192214966, "sampling/importance_sampling_ratio/max": 1.8340866565704346, "entropy": 2.569779545068741, "clip_ratio/low_mean": 0.0794306411407888, "clip_ratio/low_min": 0.0794306411407888, "clip_ratio/high_mean": 0.05716765904799104, "clip_ratio/high_max": 0.05716765904799104, "clip_ratio/region_mean": 0.13659830018877983, "reward_total_mean": 0.4390692710876465, "reward_meter_mean": 0.4516580104827881, "reward_meter_std": 0.4649323523044586, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8865073919296265, "reward_repeat_soft_std": 0.18254618346691132, "reward_judge_quality_mean": 0.38749998807907104, "reward_judge_quality_std": 0.11877349019050598, "reward_total_composite_mean": 0.4390692710876465, "reward_total_composite_std": 0.12529818713665009} {"timestamp_utc": "2026-04-13T08:04:01Z", "mode": "train", "global_step": 297, "epoch": 0.02983425414364641, "loss": 0.0847, "grad_norm": 11.290495872497559, "learning_rate": 9.103030303030304e-06, "num_tokens": 546975.0, "completions/mean_length": 57.75, "completions/min_length": 51.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.75, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.8845210075378418, "rewards/meter/std": 0.14227625727653503, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.997162938117981, "rewards/repeat_soft/std": 0.002737692091614008, "rewards/judge_quality/mean": 0.4024999737739563, "rewards/judge_quality/std": 0.16446885466575623, "rewards/total_composite/mean": 0.5384292602539062, "rewards/total_composite/std": 0.2386161834001541, "reward": 0.5384292602539062, "reward_std": 0.2386161834001541, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23331233859062195, "sampling/sampling_logp_difference/max": 2.008185863494873, "sampling/importance_sampling_ratio/min": 0.13423196971416473, "sampling/importance_sampling_ratio/mean": 1.0346351861953735, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.3425499498844147, "clip_ratio/low_mean": 0.04545697756111622, "clip_ratio/low_min": 0.04545697756111622, "clip_ratio/high_mean": 0.1748992707580328, "clip_ratio/high_max": 0.1748992707580328, "clip_ratio/region_mean": 0.22035624831914902, "reward_total_mean": 0.5384292602539062, "reward_meter_mean": 0.8845210075378418, "reward_meter_std": 0.14227625727653503, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.997162938117981, "reward_repeat_soft_std": 0.002737692091614008, "reward_judge_quality_mean": 0.4024999737739563, "reward_judge_quality_std": 0.16446885466575623, "reward_total_composite_mean": 0.5384292602539062, "reward_total_composite_std": 0.2386161834001541} {"timestamp_utc": "2026-04-13T08:04:09Z", "mode": "train", "global_step": 298, "epoch": 0.0299347061778001, "loss": 0.0433, "grad_norm": 19.058700561523438, "learning_rate": 9.100000000000001e-06, "num_tokens": 548888.0, "completions/mean_length": 50.125, "completions/min_length": 39.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.125, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.3783789873123169, "rewards/meter/std": 0.25831398367881775, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9826259613037109, "rewards/repeat_soft/std": 0.015653958544135094, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.43459272384643555, "rewards/total_composite/std": 0.08175422251224518, "reward": 0.43459272384643555, "reward_std": 0.08175422251224518, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22767361998558044, "sampling/sampling_logp_difference/max": 1.933873176574707, "sampling/importance_sampling_ratio/min": 0.14458711445331573, "sampling/importance_sampling_ratio/mean": 1.0235506296157837, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6796212941408157, "clip_ratio/low_mean": 0.10850738547742367, "clip_ratio/low_min": 0.10850738547742367, "clip_ratio/high_mean": 0.07100156880915165, "clip_ratio/high_max": 0.07100156880915165, "clip_ratio/region_mean": 0.17950895428657532, "reward_total_mean": 0.43459272384643555, "reward_meter_mean": 0.3783789873123169, "reward_meter_std": 0.25831398367881775, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9826259613037109, "reward_repeat_soft_std": 0.015653958544135094, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.43459272384643555, "reward_total_composite_std": 0.08175422251224518} {"timestamp_utc": "2026-04-13T08:04:15Z", "mode": "train", "global_step": 299, "epoch": 0.03003515821195379, "loss": 0.1628, "grad_norm": 12.718059539794922, "learning_rate": 9.096969696969698e-06, "num_tokens": 550751.0, "completions/mean_length": 60.875, "completions/min_length": 47.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.875, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.6708670258522034, "rewards/meter/std": 0.3672305643558502, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9817839860916138, "rewards/repeat_soft/std": 0.02569853514432907, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.12906256318092346, "rewards/total_composite/mean": 0.5338301658630371, "rewards/total_composite/std": 0.10163712501525879, "reward": 0.5338301658630371, "reward_std": 0.1016371101140976, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19892556965351105, "sampling/sampling_logp_difference/max": 1.3361644744873047, "sampling/importance_sampling_ratio/min": 0.26285192370414734, "sampling/importance_sampling_ratio/mean": 1.06418776512146, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.8512686491012573, "clip_ratio/low_mean": 0.061188691295683384, "clip_ratio/low_min": 0.061188691295683384, "clip_ratio/high_mean": 0.09996799007058144, "clip_ratio/high_max": 0.09996799007058144, "clip_ratio/region_mean": 0.16115668136626482, "reward_total_mean": 0.5338301658630371, "reward_meter_mean": 0.6708670258522034, "reward_meter_std": 0.3672305643558502, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9817839860916138, "reward_repeat_soft_std": 0.02569853514432907, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.12906256318092346, "reward_total_composite_mean": 0.5338301658630371, "reward_total_composite_std": 0.10163712501525879} {"timestamp_utc": "2026-04-13T08:04:22Z", "mode": "train", "global_step": 300, "epoch": 0.030135610246107485, "loss": 0.1161, "grad_norm": 18.334640502929688, "learning_rate": 9.093939393939395e-06, "num_tokens": 552255.0, "completions/mean_length": 33.0, "completions/min_length": 29.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.0, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.7040756940841675, "rewards/meter/std": 0.40580132603645325, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9876308441162109, "rewards/repeat_soft/std": 0.019401581957936287, "rewards/judge_quality/mean": 0.3962499797344208, "rewards/judge_quality/std": 0.09085899591445923, "rewards/total_composite/mean": 0.5045775175094604, "rewards/total_composite/std": 0.22371245920658112, "reward": 0.5045775175094604, "reward_std": 0.22371245920658112, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2362433224916458, "sampling/sampling_logp_difference/max": 1.2512593269348145, "sampling/importance_sampling_ratio/min": 0.2861442267894745, "sampling/importance_sampling_ratio/mean": 1.0532236099243164, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6142437160015106, "clip_ratio/low_mean": 0.06884342804551125, "clip_ratio/low_min": 0.06884342804551125, "clip_ratio/high_mean": 0.17780407890677452, "clip_ratio/high_max": 0.17780407890677452, "clip_ratio/region_mean": 0.24664750695228577, "reward_total_mean": 0.5045775175094604, "reward_meter_mean": 0.7040756940841675, "reward_meter_std": 0.40580132603645325, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9876308441162109, "reward_repeat_soft_std": 0.019401581957936287, "reward_judge_quality_mean": 0.3962499797344208, "reward_judge_quality_std": 0.09085899591445923, "reward_total_composite_mean": 0.5045775175094604, "reward_total_composite_std": 0.22371245920658112} {"timestamp_utc": "2026-04-13T08:05:13Z", "mode": "eval", "global_step": 300, "epoch": 0.030135610246107485, "eval_loss": NaN, "eval_runtime": 50.4786, "eval_samples_per_second": 1.585, "eval_steps_per_second": 0.198, "eval_num_tokens": 552255.0, "eval_completions/mean_length": 60.25, "eval_completions/min_length": 26.1, "eval_completions/max_length": 135.1, "eval_completions/clipped_ratio": 0.0125, "eval_completions/mean_terminated_length": 54.56607170104981, "eval_completions/min_terminated_length": 26.1, "eval_completions/max_terminated_length": 92.7, "eval_rewards/meter/mean": 0.4739754617214203, "eval_rewards/meter/std": 0.35049658715724946, "eval_rewards/count_adherence/mean": 0.962291669845581, "eval_rewards/count_adherence/std": 0.08423538282513618, "eval_rewards/hard_gate/mean": 0.975, "eval_rewards/hard_gate/std": 0.07071067690849304, "eval_rewards/repeat_soft/mean": 0.9883093476295471, "eval_rewards/repeat_soft/std": 0.02110612459946424, "eval_rewards/judge_quality/mean": 0.46175000071525574, "eval_rewards/judge_quality/std": 0.16253908574581147, "eval_rewards/total_composite/mean": 0.4752084851264954, "eval_rewards/total_composite/std": 0.13369336500763893, "eval_reward": 0.4752084851264954, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.15603995621204375, "eval_sampling/sampling_logp_difference/max": 1.1039558410644532, "eval_sampling/importance_sampling_ratio/min": 0.3371889188885689, "eval_sampling/importance_sampling_ratio/mean": 1.05187269449234, "eval_sampling/importance_sampling_ratio/max": 1.5952545404434204, "eval_entropy": 2.5126993536949156, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.4752084851264954, "eval_reward_meter_mean": 0.4739754617214203, "eval_reward_meter_std": 0.35049658715724946, "eval_reward_count_adherence_mean": 0.962291669845581, "eval_reward_count_adherence_std": 0.08423538282513618, "eval_reward_hard_gate_mean": 0.975, "eval_reward_hard_gate_std": 0.07071067690849304, "eval_reward_repeat_soft_mean": 0.9883093476295471, "eval_reward_repeat_soft_std": 0.02110612459946424, "eval_reward_judge_quality_mean": 0.46175000071525574, "eval_reward_judge_quality_std": 0.16253908574581147, "eval_reward_total_composite_mean": 0.4752084851264954, "eval_reward_total_composite_std": 0.13369336500763893} {"timestamp_utc": "2026-04-13T08:05:23Z", "mode": "train", "global_step": 301, "epoch": 0.030236062280261176, "loss": 0.0372, "grad_norm": 13.64366626739502, "learning_rate": 9.090909090909091e-06, "num_tokens": 554193.0, "completions/mean_length": 64.25, "completions/min_length": 57.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.25, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.2980521023273468, "rewards/meter/std": 0.3184024691581726, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9939161539077759, "rewards/repeat_soft/std": 0.009360043331980705, "rewards/judge_quality/mean": 0.3812499940395355, "rewards/judge_quality/std": 0.08166787773370743, "rewards/total_composite/mean": 0.42934858798980713, "rewards/total_composite/std": 0.08828973025083542, "reward": 0.42934858798980713, "reward_std": 0.08828973025083542, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2365809828042984, "sampling/sampling_logp_difference/max": 1.6053876876831055, "sampling/importance_sampling_ratio/min": 0.2008116990327835, "sampling/importance_sampling_ratio/mean": 1.0825746059417725, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.1526523530483246, "clip_ratio/low_mean": 0.10929449833929539, "clip_ratio/low_min": 0.10929449833929539, "clip_ratio/high_mean": 0.06331565603613853, "clip_ratio/high_max": 0.06331565603613853, "clip_ratio/region_mean": 0.17261015437543392, "reward_total_mean": 0.42934858798980713, "reward_meter_mean": 0.2980521023273468, "reward_meter_std": 0.3184024691581726, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9939161539077759, "reward_repeat_soft_std": 0.009360043331980705, "reward_judge_quality_mean": 0.3812499940395355, "reward_judge_quality_std": 0.08166787773370743, "reward_total_composite_mean": 0.42934858798980713, "reward_total_composite_std": 0.08828973025083542} {"timestamp_utc": "2026-04-13T08:05:31Z", "mode": "train", "global_step": 302, "epoch": 0.030336514314414868, "loss": 0.0041, "grad_norm": 11.884839057922363, "learning_rate": 9.087878787878788e-06, "num_tokens": 556028.0, "completions/mean_length": 58.375, "completions/min_length": 50.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.375, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.7205952405929565, "rewards/meter/std": 0.37002670764923096, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9917113780975342, "rewards/repeat_soft/std": 0.008996184915304184, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.5415973663330078, "rewards/total_composite/std": 0.10051526874303818, "reward": 0.5415973663330078, "reward_std": 0.10051526874303818, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22755461931228638, "sampling/sampling_logp_difference/max": 1.2427740097045898, "sampling/importance_sampling_ratio/min": 0.28858259320259094, "sampling/importance_sampling_ratio/mean": 1.03282630443573, "sampling/importance_sampling_ratio/max": 1.9246867895126343, "entropy": 2.900486648082733, "clip_ratio/low_mean": 0.03722166828811169, "clip_ratio/low_min": 0.03722166828811169, "clip_ratio/high_mean": 0.14813154190778732, "clip_ratio/high_max": 0.14813154190778732, "clip_ratio/region_mean": 0.185353210195899, "reward_total_mean": 0.5415973663330078, "reward_meter_mean": 0.7205952405929565, "reward_meter_std": 0.37002670764923096, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9917113780975342, "reward_repeat_soft_std": 0.008996184915304184, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.5415973663330078, "reward_total_composite_std": 0.10051526874303818} {"timestamp_utc": "2026-04-13T08:05:38Z", "mode": "train", "global_step": 303, "epoch": 0.03043696634856856, "loss": 0.0515, "grad_norm": 10.325447082519531, "learning_rate": 9.084848484848486e-06, "num_tokens": 558213.0, "completions/mean_length": 87.125, "completions/min_length": 78.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.125, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.33736515045166016, "rewards/meter/std": 0.2612042725086212, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9919346570968628, "rewards/repeat_soft/std": 0.0070076147094368935, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.4575711488723755, "rewards/total_composite/std": 0.07796068489551544, "reward": 0.4575711488723755, "reward_std": 0.07796068489551544, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20860318839550018, "sampling/sampling_logp_difference/max": 1.5033361911773682, "sampling/importance_sampling_ratio/min": 0.22238700091838837, "sampling/importance_sampling_ratio/mean": 1.048304796218872, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4992952048778534, "clip_ratio/low_mean": 0.09519331157207489, "clip_ratio/low_min": 0.09519331157207489, "clip_ratio/high_mean": 0.09584271814674139, "clip_ratio/high_max": 0.09584271814674139, "clip_ratio/region_mean": 0.19103602971881628, "reward_total_mean": 0.4575711488723755, "reward_meter_mean": 0.33736515045166016, "reward_meter_std": 0.2612042725086212, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9919346570968628, "reward_repeat_soft_std": 0.0070076147094368935, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.4575711488723755, "reward_total_composite_std": 0.07796068489551544} {"timestamp_utc": "2026-04-13T08:05:46Z", "mode": "train", "global_step": 304, "epoch": 0.03053741838272225, "loss": 0.047, "grad_norm": 11.980401992797852, "learning_rate": 9.081818181818183e-06, "num_tokens": 560301.0, "completions/mean_length": 83.0, "completions/min_length": 73.0, "completions/max_length": 100.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 83.0, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 100.0, "rewards/meter/mean": 0.3610544800758362, "rewards/meter/std": 0.272571325302124, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9927531480789185, "rewards/repeat_soft/std": 0.009791922755539417, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.4092235267162323, "rewards/total_composite/std": 0.20892708003520966, "reward": 0.4092235267162323, "reward_std": 0.20892708003520966, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22895309329032898, "sampling/sampling_logp_difference/max": 2.0566253662109375, "sampling/importance_sampling_ratio/min": 0.12788480520248413, "sampling/importance_sampling_ratio/mean": 1.0300662517547607, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2296884953975677, "clip_ratio/low_mean": 0.10625382140278816, "clip_ratio/low_min": 0.10625382140278816, "clip_ratio/high_mean": 0.07721620984375477, "clip_ratio/high_max": 0.07721620984375477, "clip_ratio/region_mean": 0.18347003124654293, "reward_total_mean": 0.4092235267162323, "reward_meter_mean": 0.3610544800758362, "reward_meter_std": 0.272571325302124, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9927531480789185, "reward_repeat_soft_std": 0.009791922755539417, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.4092235267162323, "reward_total_composite_std": 0.20892708003520966} {"timestamp_utc": "2026-04-13T08:05:54Z", "mode": "train", "global_step": 305, "epoch": 0.03063787041687594, "loss": 0.0564, "grad_norm": 19.926694869995117, "learning_rate": 9.078787878787878e-06, "num_tokens": 561755.0, "completions/mean_length": 36.75, "completions/min_length": 25.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.75, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.1998346596956253, "rewards/meter/std": 0.3175995647907257, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9959381818771362, "rewards/repeat_soft/std": 0.011488578282296658, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.17136012017726898, "rewards/total_composite/mean": 0.36549967527389526, "rewards/total_composite/std": 0.1727069914340973, "reward": 0.36549967527389526, "reward_std": 0.1727069914340973, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1940957009792328, "sampling/sampling_logp_difference/max": 1.5790209770202637, "sampling/importance_sampling_ratio/min": 0.20617684721946716, "sampling/importance_sampling_ratio/mean": 1.0141253471374512, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7443688213825226, "clip_ratio/low_mean": 0.05058035673573613, "clip_ratio/low_min": 0.05058035673573613, "clip_ratio/high_mean": 0.11338341608643532, "clip_ratio/high_max": 0.11338341608643532, "clip_ratio/region_mean": 0.16396377282217145, "reward_total_mean": 0.36549967527389526, "reward_meter_mean": 0.1998346596956253, "reward_meter_std": 0.3175995647907257, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9959381818771362, "reward_repeat_soft_std": 0.011488578282296658, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.17136012017726898, "reward_total_composite_mean": 0.36549967527389526, "reward_total_composite_std": 0.1727069914340973} {"timestamp_utc": "2026-04-13T08:06:02Z", "mode": "train", "global_step": 306, "epoch": 0.030738322451029632, "loss": -0.025, "grad_norm": 17.35762596130371, "learning_rate": 9.075757575757577e-06, "num_tokens": 563522.0, "completions/mean_length": 49.875, "completions/min_length": 41.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.875, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.5379660129547119, "rewards/meter/std": 0.389585018157959, "rewards/count_adherence/mean": 0.84375, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9965270757675171, "rewards/repeat_soft/std": 0.0026339867617934942, "rewards/judge_quality/mean": 0.3374999761581421, "rewards/judge_quality/std": 0.12291111052036285, "rewards/total_composite/mean": 0.4408089816570282, "rewards/total_composite/std": 0.11375391483306885, "reward": 0.4408089816570282, "reward_std": 0.11375392228364944, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2487066686153412, "sampling/sampling_logp_difference/max": 1.7364892959594727, "sampling/importance_sampling_ratio/min": 0.17613768577575684, "sampling/importance_sampling_ratio/mean": 1.030597448348999, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.687733069062233, "clip_ratio/low_mean": 0.1220129169523716, "clip_ratio/low_min": 0.1220129169523716, "clip_ratio/high_mean": 0.09736341051757336, "clip_ratio/high_max": 0.09736341051757336, "clip_ratio/region_mean": 0.21937632746994495, "reward_total_mean": 0.4408089816570282, "reward_meter_mean": 0.5379660129547119, "reward_meter_std": 0.389585018157959, "reward_count_adherence_mean": 0.84375, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9965270757675171, "reward_repeat_soft_std": 0.0026339867617934942, "reward_judge_quality_mean": 0.3374999761581421, "reward_judge_quality_std": 0.12291111052036285, "reward_total_composite_mean": 0.4408089816570282, "reward_total_composite_std": 0.11375391483306885} {"timestamp_utc": "2026-04-13T08:06:09Z", "mode": "train", "global_step": 307, "epoch": 0.030838774485183323, "loss": 0.0418, "grad_norm": 19.14022445678711, "learning_rate": 9.072727272727273e-06, "num_tokens": 564926.0, "completions/mean_length": 33.5, "completions/min_length": 27.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.5, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.45821669697761536, "rewards/meter/std": 0.345441997051239, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9934337139129639, "rewards/repeat_soft/std": 0.008110122755169868, "rewards/judge_quality/mean": 0.5862500071525574, "rewards/judge_quality/std": 0.2353682667016983, "rewards/total_composite/mean": 0.5238087773323059, "rewards/total_composite/std": 0.20000897347927094, "reward": 0.5238087773323059, "reward_std": 0.20000897347927094, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1961425095796585, "sampling/sampling_logp_difference/max": 1.7668094635009766, "sampling/importance_sampling_ratio/min": 0.17087732255458832, "sampling/importance_sampling_ratio/mean": 1.030051589012146, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5763475075364113, "clip_ratio/low_mean": 0.07996872556395829, "clip_ratio/low_min": 0.07996872556395829, "clip_ratio/high_mean": 0.062245048582553864, "clip_ratio/high_max": 0.062245048582553864, "clip_ratio/region_mean": 0.14221377414651215, "reward_total_mean": 0.5238087773323059, "reward_meter_mean": 0.45821669697761536, "reward_meter_std": 0.345441997051239, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9934337139129639, "reward_repeat_soft_std": 0.008110122755169868, "reward_judge_quality_mean": 0.5862500071525574, "reward_judge_quality_std": 0.2353682667016983, "reward_total_composite_mean": 0.5238087773323059, "reward_total_composite_std": 0.20000897347927094} {"timestamp_utc": "2026-04-13T08:06:16Z", "mode": "train", "global_step": 308, "epoch": 0.030939226519337018, "loss": -0.0583, "grad_norm": 16.181547164916992, "learning_rate": 9.06969696969697e-06, "num_tokens": 566707.0, "completions/mean_length": 37.625, "completions/min_length": 30.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.625, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.8855193853378296, "rewards/meter/std": 0.16866350173950195, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9974081516265869, "rewards/repeat_soft/std": 0.0032312325201928616, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.0645727664232254, "rewards/total_composite/mean": 0.5825735330581665, "rewards/total_composite/std": 0.05710341036319733, "reward": 0.5825735330581665, "reward_std": 0.057103417813777924, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21633878350257874, "sampling/sampling_logp_difference/max": 1.501877784729004, "sampling/importance_sampling_ratio/min": 0.22271157801151276, "sampling/importance_sampling_ratio/mean": 1.0522710084915161, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1090298891067505, "clip_ratio/low_mean": 0.04930555634200573, "clip_ratio/low_min": 0.04930555634200573, "clip_ratio/high_mean": 0.15730290859937668, "clip_ratio/high_max": 0.15730290859937668, "clip_ratio/region_mean": 0.2066084649413824, "reward_total_mean": 0.5825735330581665, "reward_meter_mean": 0.8855193853378296, "reward_meter_std": 0.16866350173950195, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9974081516265869, "reward_repeat_soft_std": 0.0032312325201928616, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.0645727664232254, "reward_total_composite_mean": 0.5825735330581665, "reward_total_composite_std": 0.05710341036319733} {"timestamp_utc": "2026-04-13T08:06:24Z", "mode": "train", "global_step": 309, "epoch": 0.03103967855349071, "loss": 0.0176, "grad_norm": 11.218291282653809, "learning_rate": 9.066666666666667e-06, "num_tokens": 568710.0, "completions/mean_length": 62.375, "completions/min_length": 50.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.375, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.3145693242549896, "rewards/meter/std": 0.29673534631729126, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9977501630783081, "rewards/repeat_soft/std": 0.0027243790682405233, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.19949938356876373, "rewards/total_composite/mean": 0.44601163268089294, "rewards/total_composite/std": 0.1473219394683838, "reward": 0.44601163268089294, "reward_std": 0.1473219245672226, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23919273912906647, "sampling/sampling_logp_difference/max": 1.101022720336914, "sampling/importance_sampling_ratio/min": 0.33253082633018494, "sampling/importance_sampling_ratio/mean": 1.0720115900039673, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.621856689453125, "clip_ratio/low_mean": 0.1271504582837224, "clip_ratio/low_min": 0.1271504582837224, "clip_ratio/high_mean": 0.04959016479551792, "clip_ratio/high_max": 0.04959016479551792, "clip_ratio/region_mean": 0.17674062307924032, "reward_total_mean": 0.44601163268089294, "reward_meter_mean": 0.3145693242549896, "reward_meter_std": 0.29673534631729126, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9977501630783081, "reward_repeat_soft_std": 0.0027243790682405233, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.19949938356876373, "reward_total_composite_mean": 0.44601163268089294, "reward_total_composite_std": 0.1473219394683838} {"timestamp_utc": "2026-04-13T08:06:33Z", "mode": "train", "global_step": 310, "epoch": 0.0311401305876444, "loss": 0.0655, "grad_norm": 8.005025863647461, "learning_rate": 9.063636363636365e-06, "num_tokens": 570990.0, "completions/mean_length": 107.0, "completions/min_length": 92.0, "completions/max_length": 140.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.0, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 140.0, "rewards/meter/mean": 0.8105195760726929, "rewards/meter/std": 0.17381274700164795, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.12400396168231964, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.948270320892334, "rewards/repeat_soft/std": 0.10612338781356812, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.5282342433929443, "rewards/total_composite/std": 0.09169287234544754, "reward": 0.5282342433929443, "reward_std": 0.09169287234544754, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22060352563858032, "sampling/sampling_logp_difference/max": 1.4738521575927734, "sampling/importance_sampling_ratio/min": 0.22904148697853088, "sampling/importance_sampling_ratio/mean": 1.0602737665176392, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.581006407737732, "clip_ratio/low_mean": 0.04999029543250799, "clip_ratio/low_min": 0.04999029543250799, "clip_ratio/high_mean": 0.1028110533952713, "clip_ratio/high_max": 0.1028110533952713, "clip_ratio/region_mean": 0.1528013488277793, "reward_total_mean": 0.5282342433929443, "reward_meter_mean": 0.8105195760726929, "reward_meter_std": 0.17381274700164795, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.12400396168231964, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.948270320892334, "reward_repeat_soft_std": 0.10612338781356812, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.5282342433929443, "reward_total_composite_std": 0.09169287234544754} {"timestamp_utc": "2026-04-13T08:06:41Z", "mode": "train", "global_step": 311, "epoch": 0.03124058262179809, "loss": -0.0212, "grad_norm": 13.041028022766113, "learning_rate": 9.06060606060606e-06, "num_tokens": 572897.0, "completions/mean_length": 66.375, "completions/min_length": 56.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.375, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.30376312136650085, "rewards/meter/std": 0.3042370080947876, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9922004342079163, "rewards/repeat_soft/std": 0.011386986821889877, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.10507651418447495, "rewards/total_composite/mean": 0.3983754515647888, "rewards/total_composite/std": 0.18190890550613403, "reward": 0.3983754515647888, "reward_std": 0.18190890550613403, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.25558942556381226, "sampling/sampling_logp_difference/max": 1.3604774475097656, "sampling/importance_sampling_ratio/min": 0.2565382719039917, "sampling/importance_sampling_ratio/mean": 1.054573893547058, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.934918701648712, "clip_ratio/low_mean": 0.08660714328289032, "clip_ratio/low_min": 0.08660714328289032, "clip_ratio/high_mean": 0.1264073895290494, "clip_ratio/high_max": 0.1264073895290494, "clip_ratio/region_mean": 0.21301453281193972, "reward_total_mean": 0.3983754515647888, "reward_meter_mean": 0.30376312136650085, "reward_meter_std": 0.3042370080947876, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9922004342079163, "reward_repeat_soft_std": 0.011386986821889877, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.10507651418447495, "reward_total_composite_mean": 0.3983754515647888, "reward_total_composite_std": 0.18190890550613403} {"timestamp_utc": "2026-04-13T08:06:48Z", "mode": "train", "global_step": 312, "epoch": 0.03134103465595178, "loss": 0.0742, "grad_norm": 29.928604125976562, "learning_rate": 9.057575757575759e-06, "num_tokens": 574291.0, "completions/mean_length": 20.25, "completions/min_length": 17.0, "completions/max_length": 24.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 20.25, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 24.0, "rewards/meter/mean": 0.6075140833854675, "rewards/meter/std": 0.39379948377609253, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.3462499976158142, "rewards/judge_quality/std": 0.10336308926343918, "rewards/total_composite/mean": 0.46354204416275024, "rewards/total_composite/std": 0.07323381304740906, "reward": 0.46354204416275024, "reward_std": 0.07323379814624786, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23011985421180725, "sampling/sampling_logp_difference/max": 1.4686622619628906, "sampling/importance_sampling_ratio/min": 0.23023326694965363, "sampling/importance_sampling_ratio/mean": 1.016001582145691, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.196314051747322, "clip_ratio/low_mean": 0.0753822373226285, "clip_ratio/low_min": 0.0753822373226285, "clip_ratio/high_mean": 0.1280034352093935, "clip_ratio/high_max": 0.1280034352093935, "clip_ratio/region_mean": 0.203385672532022, "reward_total_mean": 0.46354204416275024, "reward_meter_mean": 0.6075140833854675, "reward_meter_std": 0.39379948377609253, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.3462499976158142, "reward_judge_quality_std": 0.10336308926343918, "reward_total_composite_mean": 0.46354204416275024, "reward_total_composite_std": 0.07323381304740906} {"timestamp_utc": "2026-04-13T08:06:57Z", "mode": "train", "global_step": 313, "epoch": 0.03144148669010548, "loss": -0.052, "grad_norm": 12.364789962768555, "learning_rate": 9.054545454545455e-06, "num_tokens": 576479.0, "completions/mean_length": 77.5, "completions/min_length": 58.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.5, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.4774291515350342, "rewards/meter/std": 0.23937858641147614, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9925657510757446, "rewards/repeat_soft/std": 0.0036443264689296484, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4078633189201355, "rewards/total_composite/std": 0.17348188161849976, "reward": 0.4078633189201355, "reward_std": 0.17348188161849976, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21338839828968048, "sampling/sampling_logp_difference/max": 2.0070290565490723, "sampling/importance_sampling_ratio/min": 0.13438734412193298, "sampling/importance_sampling_ratio/mean": 1.0475268363952637, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.310043603181839, "clip_ratio/low_mean": 0.017241379246115685, "clip_ratio/low_min": 0.017241379246115685, "clip_ratio/high_mean": 0.19226273521780968, "clip_ratio/high_max": 0.19226273521780968, "clip_ratio/region_mean": 0.20950411446392536, "reward_total_mean": 0.4078633189201355, "reward_meter_mean": 0.4774291515350342, "reward_meter_std": 0.23937858641147614, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9925657510757446, "reward_repeat_soft_std": 0.0036443264689296484, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4078633189201355, "reward_total_composite_std": 0.17348188161849976} {"timestamp_utc": "2026-04-13T08:07:03Z", "mode": "train", "global_step": 314, "epoch": 0.031541938724259165, "loss": 0.0678, "grad_norm": 21.277273178100586, "learning_rate": 9.051515151515152e-06, "num_tokens": 577769.0, "completions/mean_length": 25.25, "completions/min_length": 17.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.25, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.9609196186065674, "rewards/meter/std": 0.09015358984470367, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.6525000333786011, "rewards/judge_quality/std": 0.2921227812767029, "rewards/total_composite/mean": 0.7461719512939453, "rewards/total_composite/std": 0.17513616383075714, "reward": 0.7461719512939453, "reward_std": 0.17513614892959595, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17423312366008759, "sampling/sampling_logp_difference/max": 1.3621187210083008, "sampling/importance_sampling_ratio/min": 0.2561175525188446, "sampling/importance_sampling_ratio/mean": 1.039286494255066, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6310700699687004, "clip_ratio/low_mean": 0.05719997314736247, "clip_ratio/low_min": 0.05719997314736247, "clip_ratio/high_mean": 0.07325160969048738, "clip_ratio/high_max": 0.07325160969048738, "clip_ratio/region_mean": 0.13045158283784986, "reward_total_mean": 0.7461719512939453, "reward_meter_mean": 0.9609196186065674, "reward_meter_std": 0.09015358984470367, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.6525000333786011, "reward_judge_quality_std": 0.2921227812767029, "reward_total_composite_mean": 0.7461719512939453, "reward_total_composite_std": 0.17513616383075714} {"timestamp_utc": "2026-04-13T08:07:11Z", "mode": "train", "global_step": 315, "epoch": 0.03164239075841286, "loss": 0.0864, "grad_norm": 20.46292495727539, "learning_rate": 9.04848484848485e-06, "num_tokens": 579251.0, "completions/mean_length": 31.25, "completions/min_length": 29.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.25, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.5624465942382812, "rewards/meter/std": 0.43712732195854187, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9856735467910767, "rewards/repeat_soft/std": 0.027919858694076538, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.5045914053916931, "rewards/total_composite/std": 0.12228681147098541, "reward": 0.5045914053916931, "reward_std": 0.12228679656982422, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2290685772895813, "sampling/sampling_logp_difference/max": 1.2177257537841797, "sampling/importance_sampling_ratio/min": 0.2959023416042328, "sampling/importance_sampling_ratio/mean": 1.0391243696212769, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4925708770751953, "clip_ratio/low_mean": 0.08946078643202782, "clip_ratio/low_min": 0.08946078643202782, "clip_ratio/high_mean": 0.0903023574501276, "clip_ratio/high_max": 0.0903023574501276, "clip_ratio/region_mean": 0.17976314388215542, "reward_total_mean": 0.5045914053916931, "reward_meter_mean": 0.5624465942382812, "reward_meter_std": 0.43712732195854187, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9856735467910767, "reward_repeat_soft_std": 0.027919858694076538, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.5045914053916931, "reward_total_composite_std": 0.12228681147098541} {"timestamp_utc": "2026-04-13T08:07:18Z", "mode": "train", "global_step": 316, "epoch": 0.03174284279256655, "loss": 0.0748, "grad_norm": 13.913999557495117, "learning_rate": 9.045454545454546e-06, "num_tokens": 580891.0, "completions/mean_length": 48.0, "completions/min_length": 37.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.0, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.3270801603794098, "rewards/meter/std": 0.34035104513168335, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9914262294769287, "rewards/repeat_soft/std": 0.011263171210885048, "rewards/judge_quality/mean": 0.5400000214576721, "rewards/judge_quality/std": 0.2218751311302185, "rewards/total_composite/mean": 0.4752548336982727, "rewards/total_composite/std": 0.16100160777568817, "reward": 0.4752548336982727, "reward_std": 0.16100160777568817, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2388077676296234, "sampling/sampling_logp_difference/max": 1.3835439682006836, "sampling/importance_sampling_ratio/min": 0.2506885528564453, "sampling/importance_sampling_ratio/mean": 1.0462722778320312, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.769525781273842, "clip_ratio/low_mean": 0.1359204314649105, "clip_ratio/low_min": 0.1359204314649105, "clip_ratio/high_mean": 0.0889878123998642, "clip_ratio/high_max": 0.0889878123998642, "clip_ratio/region_mean": 0.2249082438647747, "reward_total_mean": 0.4752548336982727, "reward_meter_mean": 0.3270801603794098, "reward_meter_std": 0.34035104513168335, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9914262294769287, "reward_repeat_soft_std": 0.011263171210885048, "reward_judge_quality_mean": 0.5400000214576721, "reward_judge_quality_std": 0.2218751311302185, "reward_total_composite_mean": 0.4752548336982727, "reward_total_composite_std": 0.16100160777568817} {"timestamp_utc": "2026-04-13T08:07:25Z", "mode": "train", "global_step": 317, "epoch": 0.03184329482672024, "loss": 0.0683, "grad_norm": 18.654348373413086, "learning_rate": 9.042424242424244e-06, "num_tokens": 582591.0, "completions/mean_length": 35.5, "completions/min_length": 29.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.5, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.6177049875259399, "rewards/meter/std": 0.4753933548927307, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9828039407730103, "rewards/repeat_soft/std": 0.04082963243126869, "rewards/judge_quality/mean": 0.4925000071525574, "rewards/judge_quality/std": 0.16446885466575623, "rewards/total_composite/mean": 0.5686860680580139, "rewards/total_composite/std": 0.18503132462501526, "reward": 0.5686860680580139, "reward_std": 0.18503132462501526, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2160700559616089, "sampling/sampling_logp_difference/max": 1.0445046424865723, "sampling/importance_sampling_ratio/min": 0.35186606645584106, "sampling/importance_sampling_ratio/mean": 1.014116883277893, "sampling/importance_sampling_ratio/max": 1.9738162755966187, "entropy": 2.671849325299263, "clip_ratio/low_mean": 0.04806985380128026, "clip_ratio/low_min": 0.04806985380128026, "clip_ratio/high_mean": 0.1266771275550127, "clip_ratio/high_max": 0.1266771275550127, "clip_ratio/region_mean": 0.17474698135629296, "reward_total_mean": 0.5686860680580139, "reward_meter_mean": 0.6177049875259399, "reward_meter_std": 0.4753933548927307, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9828039407730103, "reward_repeat_soft_std": 0.04082963243126869, "reward_judge_quality_mean": 0.4925000071525574, "reward_judge_quality_std": 0.16446885466575623, "reward_total_composite_mean": 0.5686860680580139, "reward_total_composite_std": 0.18503132462501526} {"timestamp_utc": "2026-04-13T08:07:34Z", "mode": "train", "global_step": 318, "epoch": 0.03194374686087393, "loss": 0.1333, "grad_norm": 16.658180236816406, "learning_rate": 9.03939393939394e-06, "num_tokens": 584484.0, "completions/mean_length": 68.625, "completions/min_length": 57.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.625, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.5067485570907593, "rewards/meter/std": 0.3414824604988098, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9948186874389648, "rewards/repeat_soft/std": 0.003807717002928257, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.478193461894989, "rewards/total_composite/std": 0.09166455268859863, "reward": 0.478193461894989, "reward_std": 0.09166455268859863, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.25587788224220276, "sampling/sampling_logp_difference/max": 2.545060157775879, "sampling/importance_sampling_ratio/min": 0.07846833020448685, "sampling/importance_sampling_ratio/mean": 1.0415856838226318, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1368319243192673, "clip_ratio/low_mean": 0.1427593883126974, "clip_ratio/low_min": 0.1427593883126974, "clip_ratio/high_mean": 0.09945355169475079, "clip_ratio/high_max": 0.09945355169475079, "clip_ratio/region_mean": 0.2422129400074482, "reward_total_mean": 0.478193461894989, "reward_meter_mean": 0.5067485570907593, "reward_meter_std": 0.3414824604988098, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9948186874389648, "reward_repeat_soft_std": 0.003807717002928257, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.478193461894989, "reward_total_composite_std": 0.09166455268859863} {"timestamp_utc": "2026-04-13T08:07:41Z", "mode": "train", "global_step": 319, "epoch": 0.032044198895027624, "loss": -0.0281, "grad_norm": 18.42127799987793, "learning_rate": 9.036363636363638e-06, "num_tokens": 585887.0, "completions/mean_length": 25.375, "completions/min_length": 19.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.375, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.7930670976638794, "rewards/meter/std": 0.3408731520175934, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.3799999952316284, "rewards/judge_quality/std": 0.11501552164554596, "rewards/total_composite/mean": 0.532849133014679, "rewards/total_composite/std": 0.1017293781042099, "reward": 0.532849133014679, "reward_std": 0.1017293632030487, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20987407863140106, "sampling/sampling_logp_difference/max": 1.3737754821777344, "sampling/importance_sampling_ratio/min": 0.2531493902206421, "sampling/importance_sampling_ratio/mean": 1.0605309009552002, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.9036002308130264, "clip_ratio/low_mean": 0.060653116554021835, "clip_ratio/low_min": 0.060653116554021835, "clip_ratio/high_mean": 0.06566179590299726, "clip_ratio/high_max": 0.06566179590299726, "clip_ratio/region_mean": 0.1263149124570191, "reward_total_mean": 0.532849133014679, "reward_meter_mean": 0.7930670976638794, "reward_meter_std": 0.3408731520175934, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.3799999952316284, "reward_judge_quality_std": 0.11501552164554596, "reward_total_composite_mean": 0.532849133014679, "reward_total_composite_std": 0.1017293781042099} {"timestamp_utc": "2026-04-13T08:07:48Z", "mode": "train", "global_step": 320, "epoch": 0.03214465092918132, "loss": 0.045, "grad_norm": 23.47793960571289, "learning_rate": 9.033333333333334e-06, "num_tokens": 587425.0, "completions/mean_length": 30.25, "completions/min_length": 24.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.25, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.7791892886161804, "rewards/meter/std": 0.29988420009613037, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9933795928955078, "rewards/repeat_soft/std": 0.008616972714662552, "rewards/judge_quality/mean": 0.4024999737739563, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.5498079657554626, "rewards/total_composite/std": 0.08107728511095047, "reward": 0.5498079657554626, "reward_std": 0.08107728511095047, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1847909539937973, "sampling/sampling_logp_difference/max": 2.1464381217956543, "sampling/importance_sampling_ratio/min": 0.3022194504737854, "sampling/importance_sampling_ratio/mean": 1.0040720701217651, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9304276630282402, "clip_ratio/low_mean": 0.09081439627334476, "clip_ratio/low_min": 0.09081439627334476, "clip_ratio/high_mean": 0.07184555940330029, "clip_ratio/high_max": 0.07184555940330029, "clip_ratio/region_mean": 0.16265995567664504, "reward_total_mean": 0.5498079657554626, "reward_meter_mean": 0.7791892886161804, "reward_meter_std": 0.29988420009613037, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9933795928955078, "reward_repeat_soft_std": 0.008616972714662552, "reward_judge_quality_mean": 0.4024999737739563, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.5498079657554626, "reward_total_composite_std": 0.08107728511095047} {"timestamp_utc": "2026-04-13T08:07:56Z", "mode": "train", "global_step": 321, "epoch": 0.03224510296333501, "loss": 0.0122, "grad_norm": 11.308182716369629, "learning_rate": 9.030303030303031e-06, "num_tokens": 589232.0, "completions/mean_length": 63.875, "completions/min_length": 56.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.875, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.6346712112426758, "rewards/meter/std": 0.3175996243953705, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9911579489707947, "rewards/repeat_soft/std": 0.0027391016483306885, "rewards/judge_quality/mean": 0.6825000047683716, "rewards/judge_quality/std": 0.23260943591594696, "rewards/total_composite/mean": 0.6405850648880005, "rewards/total_composite/std": 0.1904749721288681, "reward": 0.6405850648880005, "reward_std": 0.1904749721288681, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13196319341659546, "sampling/sampling_logp_difference/max": 2.581531047821045, "sampling/importance_sampling_ratio/min": 0.07565807551145554, "sampling/importance_sampling_ratio/mean": 1.0030826330184937, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5553000271320343, "clip_ratio/low_mean": 0.07165405922569335, "clip_ratio/low_min": 0.07165405922569335, "clip_ratio/high_mean": 0.0394513588398695, "clip_ratio/high_max": 0.0394513588398695, "clip_ratio/region_mean": 0.11110541806556284, "reward_total_mean": 0.6405850648880005, "reward_meter_mean": 0.6346712112426758, "reward_meter_std": 0.3175996243953705, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9911579489707947, "reward_repeat_soft_std": 0.0027391016483306885, "reward_judge_quality_mean": 0.6825000047683716, "reward_judge_quality_std": 0.23260943591594696, "reward_total_composite_mean": 0.6405850648880005, "reward_total_composite_std": 0.1904749721288681} {"timestamp_utc": "2026-04-13T08:08:04Z", "mode": "train", "global_step": 322, "epoch": 0.0323455549974887, "loss": 0.1136, "grad_norm": 10.704634666442871, "learning_rate": 9.027272727272728e-06, "num_tokens": 591174.0, "completions/mean_length": 79.75, "completions/min_length": 67.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.75, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.24984392523765564, "rewards/meter/std": 0.15510807931423187, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9884793758392334, "rewards/repeat_soft/std": 0.009767917916178703, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.41147929430007935, "rewards/total_composite/std": 0.047434739768505096, "reward": 0.41147929430007935, "reward_std": 0.047434743493795395, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24083074927330017, "sampling/sampling_logp_difference/max": 1.3724861145019531, "sampling/importance_sampling_ratio/min": 0.25347599387168884, "sampling/importance_sampling_ratio/mean": 1.0627409219741821, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.4511457979679108, "clip_ratio/low_mean": 0.11510831117630005, "clip_ratio/low_min": 0.11510831117630005, "clip_ratio/high_mean": 0.10612666979432106, "clip_ratio/high_max": 0.10612666979432106, "clip_ratio/region_mean": 0.2212349809706211, "reward_total_mean": 0.41147929430007935, "reward_meter_mean": 0.24984392523765564, "reward_meter_std": 0.15510807931423187, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9884793758392334, "reward_repeat_soft_std": 0.009767917916178703, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.41147929430007935, "reward_total_composite_std": 0.047434739768505096} {"timestamp_utc": "2026-04-13T08:08:16Z", "mode": "train", "global_step": 323, "epoch": 0.03244600703164239, "loss": -0.0703, "grad_norm": 3.8716347217559814, "learning_rate": 9.024242424242426e-06, "num_tokens": 592750.0, "completions/mean_length": 94.0, "completions/min_length": 29.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 34.28571701049805, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.5991547703742981, "rewards/meter/std": 0.3987487852573395, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9954737424850464, "rewards/repeat_soft/std": 0.006524989847093821, "rewards/judge_quality/mean": 0.44749999046325684, "rewards/judge_quality/std": 0.23407875001430511, "rewards/total_composite/mean": 0.44736433029174805, "rewards/total_composite/std": 0.22696553170681, "reward": 0.44736433029174805, "reward_std": 0.22696553170681, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22250120341777802, "sampling/sampling_logp_difference/max": 1.0450553894042969, "sampling/importance_sampling_ratio/min": 0.35167235136032104, "sampling/importance_sampling_ratio/mean": 1.0371218919754028, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9466101080179214, "clip_ratio/low_mean": 0.04711751639842987, "clip_ratio/low_min": 0.04711751639842987, "clip_ratio/high_mean": 0.14875785633921623, "clip_ratio/high_max": 0.14875785633921623, "clip_ratio/region_mean": 0.1958753727376461, "reward_total_mean": 0.44736433029174805, "reward_meter_mean": 0.5991547703742981, "reward_meter_std": 0.3987487852573395, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9954737424850464, "reward_repeat_soft_std": 0.006524989847093821, "reward_judge_quality_mean": 0.44749999046325684, "reward_judge_quality_std": 0.23407875001430511, "reward_total_composite_mean": 0.44736433029174805, "reward_total_composite_std": 0.22696553170681} {"timestamp_utc": "2026-04-13T08:08:24Z", "mode": "train", "global_step": 324, "epoch": 0.032546459065796084, "loss": 0.075, "grad_norm": 13.876107215881348, "learning_rate": 9.021212121212121e-06, "num_tokens": 594402.0, "completions/mean_length": 46.5, "completions/min_length": 41.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.5, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.5066701173782349, "rewards/meter/std": 0.32948172092437744, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9983240365982056, "rewards/repeat_soft/std": 0.0026988540776073933, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4880695343017578, "rewards/total_composite/std": 0.08996383100748062, "reward": 0.4880695343017578, "reward_std": 0.08996384590864182, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2152794450521469, "sampling/sampling_logp_difference/max": 1.6851024627685547, "sampling/importance_sampling_ratio/min": 0.18542543053627014, "sampling/importance_sampling_ratio/mean": 1.0551681518554688, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.030674993991852, "clip_ratio/low_mean": 0.10947934351861477, "clip_ratio/low_min": 0.10947934351861477, "clip_ratio/high_mean": 0.08234127052128315, "clip_ratio/high_max": 0.08234127052128315, "clip_ratio/region_mean": 0.19182061403989792, "reward_total_mean": 0.4880695343017578, "reward_meter_mean": 0.5066701173782349, "reward_meter_std": 0.32948172092437744, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9983240365982056, "reward_repeat_soft_std": 0.0026988540776073933, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4880695343017578, "reward_total_composite_std": 0.08996383100748062} {"timestamp_utc": "2026-04-13T08:08:32Z", "mode": "train", "global_step": 325, "epoch": 0.03264691109994977, "loss": 0.0071, "grad_norm": 16.47154426574707, "learning_rate": 9.01818181818182e-06, "num_tokens": 596013.0, "completions/mean_length": 38.375, "completions/min_length": 33.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.375, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.7837358713150024, "rewards/meter/std": 0.24594815075397491, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9950377941131592, "rewards/repeat_soft/std": 0.006041733082383871, "rewards/judge_quality/mean": 0.5974999666213989, "rewards/judge_quality/std": 0.220891073346138, "rewards/total_composite/mean": 0.6733725666999817, "rewards/total_composite/std": 0.19483089447021484, "reward": 0.6733725666999817, "reward_std": 0.19483089447021484, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20909790694713593, "sampling/sampling_logp_difference/max": 1.372457504272461, "sampling/importance_sampling_ratio/min": 0.25348323583602905, "sampling/importance_sampling_ratio/mean": 1.057938575744629, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6385374814271927, "clip_ratio/low_mean": 0.11629138700664043, "clip_ratio/low_min": 0.11629138700664043, "clip_ratio/high_mean": 0.05888960137963295, "clip_ratio/high_max": 0.05888960137963295, "clip_ratio/region_mean": 0.17518098838627338, "reward_total_mean": 0.6733725666999817, "reward_meter_mean": 0.7837358713150024, "reward_meter_std": 0.24594815075397491, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9950377941131592, "reward_repeat_soft_std": 0.006041733082383871, "reward_judge_quality_mean": 0.5974999666213989, "reward_judge_quality_std": 0.220891073346138, "reward_total_composite_mean": 0.6733725666999817, "reward_total_composite_std": 0.19483089447021484} {"timestamp_utc": "2026-04-13T08:08:39Z", "mode": "train", "global_step": 326, "epoch": 0.032747363134103466, "loss": 0.0723, "grad_norm": 15.22860336303711, "learning_rate": 9.015151515151516e-06, "num_tokens": 597468.0, "completions/mean_length": 37.875, "completions/min_length": 28.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.875, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.7003387212753296, "rewards/meter/std": 0.3372787833213806, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9941596388816833, "rewards/repeat_soft/std": 0.008668807335197926, "rewards/judge_quality/mean": 0.6487500071525574, "rewards/judge_quality/std": 0.2507951855659485, "rewards/total_composite/mean": 0.6733263731002808, "rewards/total_composite/std": 0.22199413180351257, "reward": 0.6733263731002808, "reward_std": 0.22199410200119019, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1748155802488327, "sampling/sampling_logp_difference/max": 1.2661004066467285, "sampling/importance_sampling_ratio/min": 0.2819288969039917, "sampling/importance_sampling_ratio/mean": 1.0314687490463257, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3604123145341873, "clip_ratio/low_mean": 0.11356301605701447, "clip_ratio/low_min": 0.11356301605701447, "clip_ratio/high_mean": 0.0538545292802155, "clip_ratio/high_max": 0.0538545292802155, "clip_ratio/region_mean": 0.16741754533722997, "reward_total_mean": 0.6733263731002808, "reward_meter_mean": 0.7003387212753296, "reward_meter_std": 0.3372787833213806, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9941596388816833, "reward_repeat_soft_std": 0.008668807335197926, "reward_judge_quality_mean": 0.6487500071525574, "reward_judge_quality_std": 0.2507951855659485, "reward_total_composite_mean": 0.6733263731002808, "reward_total_composite_std": 0.22199413180351257} {"timestamp_utc": "2026-04-13T08:08:47Z", "mode": "train", "global_step": 327, "epoch": 0.03284781516825716, "loss": 0.1348, "grad_norm": 21.305774688720703, "learning_rate": 9.012121212121213e-06, "num_tokens": 598802.0, "completions/mean_length": 30.75, "completions/min_length": 22.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.75, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.8316705226898193, "rewards/meter/std": 0.2706790566444397, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9990543127059937, "rewards/repeat_soft/std": 0.001751757226884365, "rewards/judge_quality/mean": 0.6737500429153442, "rewards/judge_quality/std": 0.263435423374176, "rewards/total_composite/mean": 0.6967610120773315, "rewards/total_composite/std": 0.17174653708934784, "reward": 0.6967610120773315, "reward_std": 0.17174653708934784, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22484582662582397, "sampling/sampling_logp_difference/max": 1.5765159130096436, "sampling/importance_sampling_ratio/min": 0.20669399201869965, "sampling/importance_sampling_ratio/mean": 1.0373828411102295, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4714982882142067, "clip_ratio/low_mean": 0.10420633293688297, "clip_ratio/low_min": 0.10420633293688297, "clip_ratio/high_mean": 0.07759773265570402, "clip_ratio/high_max": 0.07759773265570402, "clip_ratio/region_mean": 0.181804065592587, "reward_total_mean": 0.6967610120773315, "reward_meter_mean": 0.8316705226898193, "reward_meter_std": 0.2706790566444397, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9990543127059937, "reward_repeat_soft_std": 0.001751757226884365, "reward_judge_quality_mean": 0.6737500429153442, "reward_judge_quality_std": 0.263435423374176, "reward_total_composite_mean": 0.6967610120773315, "reward_total_composite_std": 0.17174653708934784} {"timestamp_utc": "2026-04-13T08:08:55Z", "mode": "train", "global_step": 328, "epoch": 0.03294826720241085, "loss": -0.1995, "grad_norm": 25.675718307495117, "learning_rate": 9.00909090909091e-06, "num_tokens": 600250.0, "completions/mean_length": 22.0, "completions/min_length": 11.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.0, "completions/min_terminated_length": 11.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.5685776472091675, "rewards/meter/std": 0.465649276971817, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.4629100561141968, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9718749523162842, "rewards/repeat_soft/std": 0.017359137535095215, "rewards/judge_quality/mean": 0.6775000095367432, "rewards/judge_quality/std": 0.25949129462242126, "rewards/total_composite/mean": 0.5208595395088196, "rewards/total_composite/std": 0.3706021010875702, "reward": 0.5208595395088196, "reward_std": 0.3706021010875702, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1883666068315506, "sampling/sampling_logp_difference/max": 1.1565308570861816, "sampling/importance_sampling_ratio/min": 0.31457558274269104, "sampling/importance_sampling_ratio/mean": 1.034841537475586, "sampling/importance_sampling_ratio/max": 1.9202556610107422, "entropy": 2.0162010490894318, "clip_ratio/low_mean": 0.02864583395421505, "clip_ratio/low_min": 0.02864583395421505, "clip_ratio/high_mean": 0.1330465618520975, "clip_ratio/high_max": 0.1330465618520975, "clip_ratio/region_mean": 0.16169239580631256, "reward_total_mean": 0.5208595395088196, "reward_meter_mean": 0.5685776472091675, "reward_meter_std": 0.465649276971817, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.4629100561141968, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9718749523162842, "reward_repeat_soft_std": 0.017359137535095215, "reward_judge_quality_mean": 0.6775000095367432, "reward_judge_quality_std": 0.25949129462242126, "reward_total_composite_mean": 0.5208595395088196, "reward_total_composite_std": 0.3706021010875702} {"timestamp_utc": "2026-04-13T08:09:02Z", "mode": "train", "global_step": 329, "epoch": 0.03304871923656454, "loss": 0.0671, "grad_norm": 23.811079025268555, "learning_rate": 9.006060606060607e-06, "num_tokens": 601833.0, "completions/mean_length": 22.875, "completions/min_length": 17.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.875, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.3704826235771179, "rewards/meter/std": 0.4448740482330322, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9671874642372131, "rewards/repeat_soft/std": 0.013258260674774647, "rewards/judge_quality/mean": 0.3687499761581421, "rewards/judge_quality/std": 0.10802611708641052, "rewards/total_composite/mean": 0.4032096266746521, "rewards/total_composite/std": 0.19858667254447937, "reward": 0.4032096266746521, "reward_std": 0.19858667254447937, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2445867955684662, "sampling/sampling_logp_difference/max": 1.0363855361938477, "sampling/importance_sampling_ratio/min": 0.3547345697879791, "sampling/importance_sampling_ratio/mean": 1.0885217189788818, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.910661995410919, "clip_ratio/low_mean": 0.11830841470509768, "clip_ratio/low_min": 0.11830841470509768, "clip_ratio/high_mean": 0.04718535486608744, "clip_ratio/high_max": 0.04718535486608744, "clip_ratio/region_mean": 0.1654937695711851, "reward_total_mean": 0.4032096266746521, "reward_meter_mean": 0.3704826235771179, "reward_meter_std": 0.4448740482330322, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9671874642372131, "reward_repeat_soft_std": 0.013258260674774647, "reward_judge_quality_mean": 0.3687499761581421, "reward_judge_quality_std": 0.10802611708641052, "reward_total_composite_mean": 0.4032096266746521, "reward_total_composite_std": 0.19858667254447937} {"timestamp_utc": "2026-04-13T08:09:10Z", "mode": "train", "global_step": 330, "epoch": 0.03314917127071823, "loss": 0.0256, "grad_norm": 17.825525283813477, "learning_rate": 9.003030303030303e-06, "num_tokens": 603503.0, "completions/mean_length": 38.75, "completions/min_length": 35.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.75, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.33915719389915466, "rewards/meter/std": 0.30123084783554077, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9892999529838562, "rewards/repeat_soft/std": 0.008622348308563232, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.4413272738456726, "rewards/total_composite/std": 0.08239845186471939, "reward": 0.4413272738456726, "reward_std": 0.0823984369635582, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1875453144311905, "sampling/sampling_logp_difference/max": 1.3236665725708008, "sampling/importance_sampling_ratio/min": 0.2661576271057129, "sampling/importance_sampling_ratio/mean": 1.0321362018585205, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4364940524101257, "clip_ratio/low_mean": 0.09573934972286224, "clip_ratio/low_min": 0.09573934972286224, "clip_ratio/high_mean": 0.08293084055185318, "clip_ratio/high_max": 0.08293084055185318, "clip_ratio/region_mean": 0.17867019027471542, "reward_total_mean": 0.4413272738456726, "reward_meter_mean": 0.33915719389915466, "reward_meter_std": 0.30123084783554077, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9892999529838562, "reward_repeat_soft_std": 0.008622348308563232, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.4413272738456726, "reward_total_composite_std": 0.08239845186471939} {"timestamp_utc": "2026-04-13T08:09:17Z", "mode": "train", "global_step": 331, "epoch": 0.033249623304871925, "loss": 0.0365, "grad_norm": 19.763568878173828, "learning_rate": 9e-06, "num_tokens": 604977.0, "completions/mean_length": 28.25, "completions/min_length": 23.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.25, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.30294397473335266, "rewards/meter/std": 0.34484946727752686, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9890438318252563, "rewards/repeat_soft/std": 0.022879350930452347, "rewards/judge_quality/mean": 0.4675000011920929, "rewards/judge_quality/std": 0.21022097766399384, "rewards/total_composite/mean": 0.4555625915527344, "rewards/total_composite/std": 0.13149835169315338, "reward": 0.4555625915527344, "reward_std": 0.1314983367919922, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24701382219791412, "sampling/sampling_logp_difference/max": 2.6028099060058594, "sampling/importance_sampling_ratio/min": 0.07406517118215561, "sampling/importance_sampling_ratio/mean": 1.0688800811767578, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4246515184640884, "clip_ratio/low_mean": 0.11589226126670837, "clip_ratio/low_min": 0.11589226126670837, "clip_ratio/high_mean": 0.078125, "clip_ratio/high_max": 0.078125, "clip_ratio/region_mean": 0.19401726126670837, "reward_total_mean": 0.4555625915527344, "reward_meter_mean": 0.30294397473335266, "reward_meter_std": 0.34484946727752686, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9890438318252563, "reward_repeat_soft_std": 0.022879350930452347, "reward_judge_quality_mean": 0.4675000011920929, "reward_judge_quality_std": 0.21022097766399384, "reward_total_composite_mean": 0.4555625915527344, "reward_total_composite_std": 0.13149835169315338} {"timestamp_utc": "2026-04-13T08:09:24Z", "mode": "train", "global_step": 332, "epoch": 0.03335007533902561, "loss": 0.054, "grad_norm": 14.886153221130371, "learning_rate": 8.996969696969697e-06, "num_tokens": 606595.0, "completions/mean_length": 34.25, "completions/min_length": 28.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.25, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.4881669580936432, "rewards/meter/std": 0.41594117879867554, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9882906079292297, "rewards/repeat_soft/std": 0.011169468984007835, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.5594117641448975, "rewards/total_composite/std": 0.23872269690036774, "reward": 0.5594117641448975, "reward_std": 0.23872269690036774, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18812359869480133, "sampling/sampling_logp_difference/max": 1.2091703414916992, "sampling/importance_sampling_ratio/min": 0.2984447777271271, "sampling/importance_sampling_ratio/mean": 1.0254172086715698, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7859932333230972, "clip_ratio/low_mean": 0.08856897242367268, "clip_ratio/low_min": 0.08856897242367268, "clip_ratio/high_mean": 0.05078725144267082, "clip_ratio/high_max": 0.05078725144267082, "clip_ratio/region_mean": 0.1393562238663435, "reward_total_mean": 0.5594117641448975, "reward_meter_mean": 0.4881669580936432, "reward_meter_std": 0.41594117879867554, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9882906079292297, "reward_repeat_soft_std": 0.011169468984007835, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.5594117641448975, "reward_total_composite_std": 0.23872269690036774} {"timestamp_utc": "2026-04-13T08:09:37Z", "mode": "train", "global_step": 333, "epoch": 0.03345052737317931, "loss": -0.064, "grad_norm": 2.7316510677337646, "learning_rate": 8.993939393939395e-06, "num_tokens": 608741.0, "completions/mean_length": 210.25, "completions/min_length": 64.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 109.66667175292969, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 212.0, "rewards/meter/mean": 0.799657940864563, "rewards/meter/std": 0.20600946247577667, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.9958186149597168, "rewards/repeat_soft/std": 0.002802198752760887, "rewards/judge_quality/mean": 0.21875, "rewards/judge_quality/std": 0.17016273736953735, "rewards/total_composite/mean": 0.333798885345459, "rewards/total_composite/std": 0.2898932695388794, "reward": 0.333798885345459, "reward_std": 0.2898932695388794, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22127531468868256, "sampling/sampling_logp_difference/max": 1.5819416046142578, "sampling/importance_sampling_ratio/min": 0.20557557046413422, "sampling/importance_sampling_ratio/mean": 1.0461658239364624, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.653821587562561, "clip_ratio/low_mean": 0.013698630034923553, "clip_ratio/low_min": 0.013698630034923553, "clip_ratio/high_mean": 0.11938321124762297, "clip_ratio/high_max": 0.11938321124762297, "clip_ratio/region_mean": 0.13308184128254652, "reward_total_mean": 0.333798885345459, "reward_meter_mean": 0.799657940864563, "reward_meter_std": 0.20600946247577667, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.9958186149597168, "reward_repeat_soft_std": 0.002802198752760887, "reward_judge_quality_mean": 0.21875, "reward_judge_quality_std": 0.17016273736953735, "reward_total_composite_mean": 0.333798885345459, "reward_total_composite_std": 0.2898932695388794} {"timestamp_utc": "2026-04-13T08:09:45Z", "mode": "train", "global_step": 334, "epoch": 0.033550979407332995, "loss": 0.0524, "grad_norm": 13.678813934326172, "learning_rate": 8.990909090909092e-06, "num_tokens": 610686.0, "completions/mean_length": 74.125, "completions/min_length": 58.0, "completions/max_length": 92.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.125, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 92.0, "rewards/meter/mean": 0.3780759572982788, "rewards/meter/std": 0.31160059571266174, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9950718879699707, "rewards/repeat_soft/std": 0.006842370145022869, "rewards/judge_quality/mean": 0.41499999165534973, "rewards/judge_quality/std": 0.14520922303199768, "rewards/total_composite/mean": 0.3830219507217407, "rewards/total_composite/std": 0.17284251749515533, "reward": 0.3830219507217407, "reward_std": 0.17284250259399414, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23534409701824188, "sampling/sampling_logp_difference/max": 1.1867132186889648, "sampling/importance_sampling_ratio/min": 0.3052228093147278, "sampling/importance_sampling_ratio/mean": 1.060614824295044, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.6963296830654144, "clip_ratio/low_mean": 0.061289879493415356, "clip_ratio/low_min": 0.061289879493415356, "clip_ratio/high_mean": 0.11982089839875698, "clip_ratio/high_max": 0.11982089839875698, "clip_ratio/region_mean": 0.18111077789217234, "reward_total_mean": 0.3830219507217407, "reward_meter_mean": 0.3780759572982788, "reward_meter_std": 0.31160059571266174, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9950718879699707, "reward_repeat_soft_std": 0.006842370145022869, "reward_judge_quality_mean": 0.41499999165534973, "reward_judge_quality_std": 0.14520922303199768, "reward_total_composite_mean": 0.3830219507217407, "reward_total_composite_std": 0.17284251749515533} {"timestamp_utc": "2026-04-13T08:09:52Z", "mode": "train", "global_step": 335, "epoch": 0.03365143144148669, "loss": 0.0727, "grad_norm": 15.958833694458008, "learning_rate": 8.98787878787879e-06, "num_tokens": 612672.0, "completions/mean_length": 63.25, "completions/min_length": 57.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.25, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.35783955454826355, "rewards/meter/std": 0.2124491184949875, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.1035098284482956, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9926632642745972, "rewards/repeat_soft/std": 0.005155050661414862, "rewards/judge_quality/mean": 0.6074999570846558, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.46032220125198364, "rewards/total_composite/std": 0.06749068945646286, "reward": 0.46032220125198364, "reward_std": 0.06749069690704346, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19007080793380737, "sampling/sampling_logp_difference/max": 1.824625849723816, "sampling/importance_sampling_ratio/min": 0.16127797961235046, "sampling/importance_sampling_ratio/mean": 1.0112042427062988, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5455197468400002, "clip_ratio/low_mean": 0.044313523918390274, "clip_ratio/low_min": 0.044313523918390274, "clip_ratio/high_mean": 0.13582287728786469, "clip_ratio/high_max": 0.13582287728786469, "clip_ratio/region_mean": 0.18013640120625496, "reward_total_mean": 0.46032220125198364, "reward_meter_mean": 0.35783955454826355, "reward_meter_std": 0.2124491184949875, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.1035098284482956, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9926632642745972, "reward_repeat_soft_std": 0.005155050661414862, "reward_judge_quality_mean": 0.6074999570846558, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.46032220125198364, "reward_total_composite_std": 0.06749068945646286} {"timestamp_utc": "2026-04-13T08:09:59Z", "mode": "train", "global_step": 336, "epoch": 0.033751883475640385, "loss": 0.0209, "grad_norm": 19.562091827392578, "learning_rate": 8.984848484848485e-06, "num_tokens": 614193.0, "completions/mean_length": 34.125, "completions/min_length": 25.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.125, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.5130771398544312, "rewards/meter/std": 0.39404621720314026, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9882612228393555, "rewards/repeat_soft/std": 0.009143383242189884, "rewards/judge_quality/mean": 0.7987500429153442, "rewards/judge_quality/std": 0.22465451061725616, "rewards/total_composite/mean": 0.6394940614700317, "rewards/total_composite/std": 0.2444366216659546, "reward": 0.6394940614700317, "reward_std": 0.2444366216659546, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1709323227405548, "sampling/sampling_logp_difference/max": 3.006063938140869, "sampling/importance_sampling_ratio/min": 0.04948607459664345, "sampling/importance_sampling_ratio/mean": 1.0001240968704224, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0319463834166527, "clip_ratio/low_mean": 0.09706691186875105, "clip_ratio/low_min": 0.09706691186875105, "clip_ratio/high_mean": 0.08584709116257727, "clip_ratio/high_max": 0.08584709116257727, "clip_ratio/region_mean": 0.18291400303132832, "reward_total_mean": 0.6394940614700317, "reward_meter_mean": 0.5130771398544312, "reward_meter_std": 0.39404621720314026, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9882612228393555, "reward_repeat_soft_std": 0.009143383242189884, "reward_judge_quality_mean": 0.7987500429153442, "reward_judge_quality_std": 0.22465451061725616, "reward_total_composite_mean": 0.6394940614700317, "reward_total_composite_std": 0.2444366216659546} {"timestamp_utc": "2026-04-13T08:10:07Z", "mode": "train", "global_step": 337, "epoch": 0.03385233550979407, "loss": 0.018, "grad_norm": 11.302577018737793, "learning_rate": 8.981818181818182e-06, "num_tokens": 616565.0, "completions/mean_length": 96.5, "completions/min_length": 83.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.5, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.22483281791210175, "rewards/meter/std": 0.31735825538635254, "rewards/count_adherence/mean": 0.8500000238418579, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9898207187652588, "rewards/repeat_soft/std": 0.006588577758520842, "rewards/judge_quality/mean": 0.3725000023841858, "rewards/judge_quality/std": 0.16368524730205536, "rewards/total_composite/mean": 0.3557148277759552, "rewards/total_composite/std": 0.2066221833229065, "reward": 0.3557148277759552, "reward_std": 0.2066221833229065, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23147423565387726, "sampling/sampling_logp_difference/max": 1.458292007446289, "sampling/importance_sampling_ratio/min": 0.23263327777385712, "sampling/importance_sampling_ratio/mean": 1.0710952281951904, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.368072062730789, "clip_ratio/low_mean": 0.10687562357634306, "clip_ratio/low_min": 0.10687562357634306, "clip_ratio/high_mean": 0.07749764993786812, "clip_ratio/high_max": 0.07749764993786812, "clip_ratio/region_mean": 0.18437327351421118, "reward_total_mean": 0.3557148277759552, "reward_meter_mean": 0.22483281791210175, "reward_meter_std": 0.31735825538635254, "reward_count_adherence_mean": 0.8500000238418579, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9898207187652588, "reward_repeat_soft_std": 0.006588577758520842, "reward_judge_quality_mean": 0.3725000023841858, "reward_judge_quality_std": 0.16368524730205536, "reward_total_composite_mean": 0.3557148277759552, "reward_total_composite_std": 0.2066221833229065} {"timestamp_utc": "2026-04-13T08:10:20Z", "mode": "train", "global_step": 338, "epoch": 0.03395278754394777, "loss": -0.1492, "grad_norm": 2.613356113433838, "learning_rate": 8.97878787878788e-06, "num_tokens": 618287.0, "completions/mean_length": 108.25, "completions/min_length": 45.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 50.57143020629883, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.8069460391998291, "rewards/meter/std": 0.3264923691749573, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.2357022762298584, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9952407479286194, "rewards/repeat_soft/std": 0.004081881605088711, "rewards/judge_quality/mean": 0.33124998211860657, "rewards/judge_quality/std": 0.13715866208076477, "rewards/total_composite/mean": 0.498996376991272, "rewards/total_composite/std": 0.2086392194032669, "reward": 0.498996376991272, "reward_std": 0.2086392045021057, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23468974232673645, "sampling/sampling_logp_difference/max": 1.2172784805297852, "sampling/importance_sampling_ratio/min": 0.29603472352027893, "sampling/importance_sampling_ratio/mean": 1.0711236000061035, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.9825278222560883, "clip_ratio/low_mean": 0.014705882407724857, "clip_ratio/low_min": 0.014705882407724857, "clip_ratio/high_mean": 0.14897500537335873, "clip_ratio/high_max": 0.14897500537335873, "clip_ratio/region_mean": 0.16368088778108358, "reward_total_mean": 0.498996376991272, "reward_meter_mean": 0.8069460391998291, "reward_meter_std": 0.3264923691749573, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.2357022762298584, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9952407479286194, "reward_repeat_soft_std": 0.004081881605088711, "reward_judge_quality_mean": 0.33124998211860657, "reward_judge_quality_std": 0.13715866208076477, "reward_total_composite_mean": 0.498996376991272, "reward_total_composite_std": 0.2086392194032669} {"timestamp_utc": "2026-04-13T08:10:29Z", "mode": "train", "global_step": 339, "epoch": 0.034053239578101455, "loss": 0.4037, "grad_norm": 11.293525695800781, "learning_rate": 8.975757575757577e-06, "num_tokens": 620037.0, "completions/mean_length": 58.75, "completions/min_length": 32.0, "completions/max_length": 233.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.75, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 233.0, "rewards/meter/mean": 0.47527629137039185, "rewards/meter/std": 0.39625003933906555, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9971331357955933, "rewards/repeat_soft/std": 0.004550741985440254, "rewards/judge_quality/mean": 0.29249998927116394, "rewards/judge_quality/std": 0.13274572789669037, "rewards/total_composite/mean": 0.460955947637558, "rewards/total_composite/std": 0.1209416389465332, "reward": 0.460955947637558, "reward_std": 0.1209416389465332, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20768223702907562, "sampling/sampling_logp_difference/max": 1.2325115203857422, "sampling/importance_sampling_ratio/min": 0.2915593981742859, "sampling/importance_sampling_ratio/mean": 1.0328776836395264, "sampling/importance_sampling_ratio/max": 1.9464133977890015, "entropy": 3.656894564628601, "clip_ratio/low_mean": 0.09924108954146504, "clip_ratio/low_min": 0.09924108954146504, "clip_ratio/high_mean": 0.0744554940611124, "clip_ratio/high_max": 0.0744554940611124, "clip_ratio/region_mean": 0.17369658360257745, "reward_total_mean": 0.460955947637558, "reward_meter_mean": 0.47527629137039185, "reward_meter_std": 0.39625003933906555, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9971331357955933, "reward_repeat_soft_std": 0.004550741985440254, "reward_judge_quality_mean": 0.29249998927116394, "reward_judge_quality_std": 0.13274572789669037, "reward_total_composite_mean": 0.460955947637558, "reward_total_composite_std": 0.1209416389465332} {"timestamp_utc": "2026-04-13T08:10:37Z", "mode": "train", "global_step": 340, "epoch": 0.03415369161225515, "loss": 0.0406, "grad_norm": 17.970109939575195, "learning_rate": 8.972727272727272e-06, "num_tokens": 621493.0, "completions/mean_length": 33.0, "completions/min_length": 29.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.0, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.7923741340637207, "rewards/meter/std": 0.34158453345298767, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9971886873245239, "rewards/repeat_soft/std": 0.003208628622815013, "rewards/judge_quality/mean": 0.6150000095367432, "rewards/judge_quality/std": 0.2655452787876129, "rewards/total_composite/mean": 0.6965111494064331, "rewards/total_composite/std": 0.20136895775794983, "reward": 0.6965111494064331, "reward_std": 0.20136895775794983, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16407504677772522, "sampling/sampling_logp_difference/max": 1.1980657577514648, "sampling/importance_sampling_ratio/min": 0.30177733302116394, "sampling/importance_sampling_ratio/mean": 1.0395454168319702, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.998213142156601, "clip_ratio/low_mean": 0.07138069625943899, "clip_ratio/low_min": 0.07138069625943899, "clip_ratio/high_mean": 0.0566600551828742, "clip_ratio/high_max": 0.0566600551828742, "clip_ratio/region_mean": 0.1280407514423132, "reward_total_mean": 0.6965111494064331, "reward_meter_mean": 0.7923741340637207, "reward_meter_std": 0.34158453345298767, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9971886873245239, "reward_repeat_soft_std": 0.003208628622815013, "reward_judge_quality_mean": 0.6150000095367432, "reward_judge_quality_std": 0.2655452787876129, "reward_total_composite_mean": 0.6965111494064331, "reward_total_composite_std": 0.20136895775794983} {"timestamp_utc": "2026-04-13T08:10:44Z", "mode": "train", "global_step": 341, "epoch": 0.03425414364640884, "loss": -0.2348, "grad_norm": 13.813849449157715, "learning_rate": 8.969696969696971e-06, "num_tokens": 623318.0, "completions/mean_length": 54.125, "completions/min_length": 36.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.125, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.6467094421386719, "rewards/meter/std": 0.24001756310462952, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9904347658157349, "rewards/repeat_soft/std": 0.006750911008566618, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.516783595085144, "rewards/total_composite/std": 0.06690096855163574, "reward": 0.516783595085144, "reward_std": 0.06690097600221634, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2130843698978424, "sampling/sampling_logp_difference/max": 0.9128122329711914, "sampling/importance_sampling_ratio/min": 0.4013938307762146, "sampling/importance_sampling_ratio/mean": 1.0654175281524658, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.926282674074173, "clip_ratio/low_mean": 0.061325354501605034, "clip_ratio/low_min": 0.061325354501605034, "clip_ratio/high_mean": 0.1286101434379816, "clip_ratio/high_max": 0.1286101434379816, "clip_ratio/region_mean": 0.18993549793958664, "reward_total_mean": 0.516783595085144, "reward_meter_mean": 0.6467094421386719, "reward_meter_std": 0.24001756310462952, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9904347658157349, "reward_repeat_soft_std": 0.006750911008566618, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.516783595085144, "reward_total_composite_std": 0.06690096855163574} {"timestamp_utc": "2026-04-13T08:10:56Z", "mode": "train", "global_step": 342, "epoch": 0.03435459568056253, "loss": 0.3122, "grad_norm": 4.728224754333496, "learning_rate": 8.966666666666667e-06, "num_tokens": 626172.0, "completions/mean_length": 176.75, "completions/min_length": 66.0, "completions/max_length": 488.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 176.75, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 488.0, "rewards/meter/mean": 0.2066187709569931, "rewards/meter/std": 0.3268250823020935, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.13363061845302582, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9973129034042358, "rewards/repeat_soft/std": 0.002697502262890339, "rewards/judge_quality/mean": 0.26375001668930054, "rewards/judge_quality/std": 0.2464570552110672, "rewards/total_composite/mean": 0.39993906021118164, "rewards/total_composite/std": 0.17000128328800201, "reward": 0.39993906021118164, "reward_std": 0.1700012981891632, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21022287011146545, "sampling/sampling_logp_difference/max": 2.8367786407470703, "sampling/importance_sampling_ratio/min": 0.058614179491996765, "sampling/importance_sampling_ratio/mean": 1.0411263704299927, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 5.513430565595627, "clip_ratio/low_mean": 0.07977114617824554, "clip_ratio/low_min": 0.07977114617824554, "clip_ratio/high_mean": 0.04984848573803902, "clip_ratio/high_max": 0.04984848573803902, "clip_ratio/region_mean": 0.12961963191628456, "reward_total_mean": 0.39993906021118164, "reward_meter_mean": 0.2066187709569931, "reward_meter_std": 0.3268250823020935, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.13363061845302582, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9973129034042358, "reward_repeat_soft_std": 0.002697502262890339, "reward_judge_quality_mean": 0.26375001668930054, "reward_judge_quality_std": 0.2464570552110672, "reward_total_composite_mean": 0.39993906021118164, "reward_total_composite_std": 0.17000128328800201} {"timestamp_utc": "2026-04-13T08:11:04Z", "mode": "train", "global_step": 343, "epoch": 0.034455047714716226, "loss": 0.027, "grad_norm": 24.89185905456543, "learning_rate": 8.963636363636364e-06, "num_tokens": 627570.0, "completions/mean_length": 20.75, "completions/min_length": 17.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 20.75, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.694798469543457, "rewards/meter/std": 0.35779568552970886, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9567934274673462, "rewards/repeat_soft/std": 0.016140474006533623, "rewards/judge_quality/mean": 0.5562499761581421, "rewards/judge_quality/std": 0.2249404788017273, "rewards/total_composite/mean": 0.5756093263626099, "rewards/total_composite/std": 0.15500064194202423, "reward": 0.5756093263626099, "reward_std": 0.15500065684318542, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20464152097702026, "sampling/sampling_logp_difference/max": 1.7080192565917969, "sampling/importance_sampling_ratio/min": 0.18122439086437225, "sampling/importance_sampling_ratio/mean": 1.0239838361740112, "sampling/importance_sampling_ratio/max": 1.7914179563522339, "entropy": 2.4688137769699097, "clip_ratio/low_mean": 0.04615705972537398, "clip_ratio/low_min": 0.04615705972537398, "clip_ratio/high_mean": 0.1430090507492423, "clip_ratio/high_max": 0.1430090507492423, "clip_ratio/region_mean": 0.1891661104746163, "reward_total_mean": 0.5756093263626099, "reward_meter_mean": 0.694798469543457, "reward_meter_std": 0.35779568552970886, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9567934274673462, "reward_repeat_soft_std": 0.016140474006533623, "reward_judge_quality_mean": 0.5562499761581421, "reward_judge_quality_std": 0.2249404788017273, "reward_total_composite_mean": 0.5756093263626099, "reward_total_composite_std": 0.15500064194202423} {"timestamp_utc": "2026-04-13T08:11:15Z", "mode": "train", "global_step": 344, "epoch": 0.034555499748869914, "loss": 0.0324, "grad_norm": 15.108419418334961, "learning_rate": 8.960606060606061e-06, "num_tokens": 629196.0, "completions/mean_length": 38.25, "completions/min_length": 29.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.25, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.6917316317558289, "rewards/meter/std": 0.3612944185733795, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9986319541931152, "rewards/repeat_soft/std": 0.0038694811519235373, "rewards/judge_quality/mean": 0.59375, "rewards/judge_quality/std": 0.22398583590984344, "rewards/total_composite/mean": 0.6348282098770142, "rewards/total_composite/std": 0.19019661843776703, "reward": 0.6348282098770142, "reward_std": 0.19019660353660583, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21258258819580078, "sampling/sampling_logp_difference/max": 1.3830804824829102, "sampling/importance_sampling_ratio/min": 0.2508047819137573, "sampling/importance_sampling_ratio/mean": 1.0417386293411255, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1024929881095886, "clip_ratio/low_mean": 0.10580164100974798, "clip_ratio/low_min": 0.10580164100974798, "clip_ratio/high_mean": 0.07892816513776779, "clip_ratio/high_max": 0.07892816513776779, "clip_ratio/region_mean": 0.18472980614751577, "reward_total_mean": 0.6348282098770142, "reward_meter_mean": 0.6917316317558289, "reward_meter_std": 0.3612944185733795, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9986319541931152, "reward_repeat_soft_std": 0.0038694811519235373, "reward_judge_quality_mean": 0.59375, "reward_judge_quality_std": 0.22398583590984344, "reward_total_composite_mean": 0.6348282098770142, "reward_total_composite_std": 0.19019661843776703} {"timestamp_utc": "2026-04-13T08:11:23Z", "mode": "train", "global_step": 345, "epoch": 0.03465595178302361, "loss": 0.1219, "grad_norm": 16.092153549194336, "learning_rate": 8.957575757575758e-06, "num_tokens": 630735.0, "completions/mean_length": 35.375, "completions/min_length": 24.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.375, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.402677983045578, "rewards/meter/std": 0.3711997866630554, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9928983449935913, "rewards/repeat_soft/std": 0.00860940758138895, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.1524970829486847, "rewards/total_composite/mean": 0.44169631600379944, "rewards/total_composite/std": 0.08157047629356384, "reward": 0.44169631600379944, "reward_std": 0.08157046884298325, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23123355209827423, "sampling/sampling_logp_difference/max": 1.5154838562011719, "sampling/importance_sampling_ratio/min": 0.2197018563747406, "sampling/importance_sampling_ratio/mean": 1.0942494869232178, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.8523354530334473, "clip_ratio/low_mean": 0.11519742850214243, "clip_ratio/low_min": 0.11519742850214243, "clip_ratio/high_mean": 0.08987254276871681, "clip_ratio/high_max": 0.08987254276871681, "clip_ratio/region_mean": 0.20506997127085924, "reward_total_mean": 0.44169631600379944, "reward_meter_mean": 0.402677983045578, "reward_meter_std": 0.3711997866630554, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9928983449935913, "reward_repeat_soft_std": 0.00860940758138895, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.1524970829486847, "reward_total_composite_mean": 0.44169631600379944, "reward_total_composite_std": 0.08157047629356384} {"timestamp_utc": "2026-04-13T08:11:34Z", "mode": "train", "global_step": 346, "epoch": 0.034756403817177296, "loss": 0.1177, "grad_norm": 17.083423614501953, "learning_rate": 8.954545454545456e-06, "num_tokens": 632355.0, "completions/mean_length": 48.5, "completions/min_length": 40.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.5, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.7290892601013184, "rewards/meter/std": 0.23382572829723358, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9975104331970215, "rewards/repeat_soft/std": 0.0042014517821371555, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.5872535705566406, "rewards/total_composite/std": 0.1103053092956543, "reward": 0.5872535705566406, "reward_std": 0.1103053092956543, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24400284886360168, "sampling/sampling_logp_difference/max": 1.9545016288757324, "sampling/importance_sampling_ratio/min": 0.14163504540920258, "sampling/importance_sampling_ratio/mean": 1.0451273918151855, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.0295895636081696, "clip_ratio/low_mean": 0.11783511005342007, "clip_ratio/low_min": 0.11783511005342007, "clip_ratio/high_mean": 0.08385869488120079, "clip_ratio/high_max": 0.08385869488120079, "clip_ratio/region_mean": 0.20169380493462086, "reward_total_mean": 0.5872535705566406, "reward_meter_mean": 0.7290892601013184, "reward_meter_std": 0.23382572829723358, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9975104331970215, "reward_repeat_soft_std": 0.0042014517821371555, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.5872535705566406, "reward_total_composite_std": 0.1103053092956543} {"timestamp_utc": "2026-04-13T08:11:48Z", "mode": "train", "global_step": 347, "epoch": 0.03485685585133099, "loss": -0.0959, "grad_norm": 2.079007863998413, "learning_rate": 8.951515151515153e-06, "num_tokens": 634374.0, "completions/mean_length": 406.375, "completions/min_length": 44.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.75, "completions/mean_terminated_length": 89.5, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 135.0, "rewards/meter/mean": 0.40235579013824463, "rewards/meter/std": 0.2786467671394348, "rewards/count_adherence/mean": 0.5, "rewards/count_adherence/std": 0.2519763112068176, "rewards/hard_gate/mean": 0.375, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.9947296380996704, "rewards/repeat_soft/std": 0.013047036714851856, "rewards/judge_quality/mean": 0.07500000298023224, "rewards/judge_quality/std": 0.0707106813788414, "rewards/total_composite/mean": 0.1072046309709549, "rewards/total_composite/std": 0.16362407803535461, "reward": 0.1072046309709549, "reward_std": 0.16362407803535461, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2555040121078491, "sampling/sampling_logp_difference/max": 1.3137226104736328, "sampling/importance_sampling_ratio/min": 0.2688175141811371, "sampling/importance_sampling_ratio/mean": 1.0996003150939941, "sampling/importance_sampling_ratio/max": 1.9200503826141357, "entropy": 1.4671365022659302, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.04957912489771843, "clip_ratio/high_max": 0.04957912489771843, "clip_ratio/region_mean": 0.04957912489771843, "reward_total_mean": 0.1072046309709549, "reward_meter_mean": 0.40235579013824463, "reward_meter_std": 0.2786467671394348, "reward_count_adherence_mean": 0.5, "reward_count_adherence_std": 0.2519763112068176, "reward_hard_gate_mean": 0.375, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.9947296380996704, "reward_repeat_soft_std": 0.013047036714851856, "reward_judge_quality_mean": 0.07500000298023224, "reward_judge_quality_std": 0.0707106813788414, "reward_total_composite_mean": 0.1072046309709549, "reward_total_composite_std": 0.16362407803535461} {"timestamp_utc": "2026-04-13T08:11:56Z", "mode": "train", "global_step": 348, "epoch": 0.03495730788548468, "loss": -0.0053, "grad_norm": 15.67885684967041, "learning_rate": 8.94848484848485e-06, "num_tokens": 635801.0, "completions/mean_length": 35.375, "completions/min_length": 31.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.375, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.22712120413780212, "rewards/meter/std": 0.3341773748397827, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9984456896781921, "rewards/repeat_soft/std": 0.00291926390491426, "rewards/judge_quality/mean": 0.5024999976158142, "rewards/judge_quality/std": 0.1348809152841568, "rewards/total_composite/mean": 0.4229438900947571, "rewards/total_composite/std": 0.09702915698289871, "reward": 0.4229438900947571, "reward_std": 0.09702915698289871, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2014777958393097, "sampling/sampling_logp_difference/max": 1.14703369140625, "sampling/importance_sampling_ratio/min": 0.31757739186286926, "sampling/importance_sampling_ratio/mean": 1.0919189453125, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.308030426502228, "clip_ratio/low_mean": 0.10102496016770601, "clip_ratio/low_min": 0.10102496016770601, "clip_ratio/high_mean": 0.056480538100004196, "clip_ratio/high_max": 0.056480538100004196, "clip_ratio/region_mean": 0.1575054982677102, "reward_total_mean": 0.4229438900947571, "reward_meter_mean": 0.22712120413780212, "reward_meter_std": 0.3341773748397827, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9984456896781921, "reward_repeat_soft_std": 0.00291926390491426, "reward_judge_quality_mean": 0.5024999976158142, "reward_judge_quality_std": 0.1348809152841568, "reward_total_composite_mean": 0.4229438900947571, "reward_total_composite_std": 0.09702915698289871} {"timestamp_utc": "2026-04-13T08:12:04Z", "mode": "train", "global_step": 349, "epoch": 0.03505775991963837, "loss": -0.0018, "grad_norm": 22.244646072387695, "learning_rate": 8.945454545454546e-06, "num_tokens": 637409.0, "completions/mean_length": 38.0, "completions/min_length": 35.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.9537421464920044, "rewards/meter/std": 0.07749338448047638, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9891766309738159, "rewards/repeat_soft/std": 0.016941606998443604, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.6537644863128662, "rewards/total_composite/std": 0.11940761655569077, "reward": 0.6537644863128662, "reward_std": 0.11940759420394897, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21780113875865936, "sampling/sampling_logp_difference/max": 1.3470020294189453, "sampling/importance_sampling_ratio/min": 0.26001861691474915, "sampling/importance_sampling_ratio/mean": 1.0623605251312256, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.431287869811058, "clip_ratio/low_mean": 0.12143311556428671, "clip_ratio/low_min": 0.12143311556428671, "clip_ratio/high_mean": 0.02631578966975212, "clip_ratio/high_max": 0.02631578966975212, "clip_ratio/region_mean": 0.14774890523403883, "reward_total_mean": 0.6537644863128662, "reward_meter_mean": 0.9537421464920044, "reward_meter_std": 0.07749338448047638, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9891766309738159, "reward_repeat_soft_std": 0.016941606998443604, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.6537644863128662, "reward_total_composite_std": 0.11940761655569077} {"timestamp_utc": "2026-04-13T08:12:12Z", "mode": "train", "global_step": 350, "epoch": 0.03515821195379206, "loss": 0.1051, "grad_norm": 17.096710205078125, "learning_rate": 8.942424242424243e-06, "num_tokens": 639036.0, "completions/mean_length": 35.375, "completions/min_length": 29.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.375, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.4493858218193054, "rewards/meter/std": 0.3458157181739807, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9930947422981262, "rewards/repeat_soft/std": 0.012191902846097946, "rewards/judge_quality/mean": 0.45875000953674316, "rewards/judge_quality/std": 0.20711885392665863, "rewards/total_composite/mean": 0.4564903974533081, "rewards/total_composite/std": 0.07396651059389114, "reward": 0.4564903974533081, "reward_std": 0.07396651059389114, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.25389793515205383, "sampling/sampling_logp_difference/max": 2.7773337364196777, "sampling/importance_sampling_ratio/min": 0.06220414116978645, "sampling/importance_sampling_ratio/mean": 1.0388528108596802, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.771635800600052, "clip_ratio/low_mean": 0.12376289814710617, "clip_ratio/low_min": 0.12376289814710617, "clip_ratio/high_mean": 0.10631196945905685, "clip_ratio/high_max": 0.10631196945905685, "clip_ratio/region_mean": 0.23007486760616302, "reward_total_mean": 0.4564903974533081, "reward_meter_mean": 0.4493858218193054, "reward_meter_std": 0.3458157181739807, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9930947422981262, "reward_repeat_soft_std": 0.012191902846097946, "reward_judge_quality_mean": 0.45875000953674316, "reward_judge_quality_std": 0.20711885392665863, "reward_total_composite_mean": 0.4564903974533081, "reward_total_composite_std": 0.07396651059389114} {"timestamp_utc": "2026-04-13T08:13:35Z", "mode": "eval", "global_step": 350, "epoch": 0.03515821195379206, "eval_loss": NaN, "eval_runtime": 83.3587, "eval_samples_per_second": 0.96, "eval_steps_per_second": 0.12, "eval_num_tokens": 639036.0, "eval_completions/mean_length": 84.8875, "eval_completions/min_length": 25.6, "eval_completions/max_length": 246.1, "eval_completions/clipped_ratio": 0.05, "eval_completions/mean_terminated_length": 62.49464378356934, "eval_completions/min_terminated_length": 25.6, "eval_completions/max_terminated_length": 133.1, "eval_rewards/meter/mean": 0.534323850274086, "eval_rewards/meter/std": 0.3636632800102234, "eval_rewards/count_adherence/mean": 0.9616666615009308, "eval_rewards/count_adherence/std": 0.08326617777347564, "eval_rewards/hard_gate/mean": 0.875, "eval_rewards/hard_gate/std": 0.25587469935417173, "eval_rewards/repeat_soft/mean": 0.9886297285556793, "eval_rewards/repeat_soft/std": 0.01991980553139001, "eval_rewards/judge_quality/mean": 0.4411250025033951, "eval_rewards/judge_quality/std": 0.19366998383775352, "eval_rewards/total_composite/mean": 0.46733091175556185, "eval_rewards/total_composite/std": 0.22581544667482376, "eval_reward": 0.46733091175556185, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.17469898462295533, "eval_sampling/sampling_logp_difference/max": 1.1589813232421875, "eval_sampling/importance_sampling_ratio/min": 0.32134189903736116, "eval_sampling/importance_sampling_ratio/mean": 1.05313880443573, "eval_sampling/importance_sampling_ratio/max": 1.7286567568778992, "eval_entropy": 3.2651810169219972, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.46733091175556185, "eval_reward_meter_mean": 0.534323850274086, "eval_reward_meter_std": 0.3636632800102234, "eval_reward_count_adherence_mean": 0.9616666615009308, "eval_reward_count_adherence_std": 0.08326617777347564, "eval_reward_hard_gate_mean": 0.875, "eval_reward_hard_gate_std": 0.25587469935417173, "eval_reward_repeat_soft_mean": 0.9886297285556793, "eval_reward_repeat_soft_std": 0.01991980553139001, "eval_reward_judge_quality_mean": 0.4411250025033951, "eval_reward_judge_quality_std": 0.19366998383775352, "eval_reward_total_composite_mean": 0.46733091175556185, "eval_reward_total_composite_std": 0.22581544667482376} {"timestamp_utc": "2026-04-13T08:13:54Z", "mode": "train", "global_step": 351, "epoch": 0.035258663987945756, "loss": -0.0436, "grad_norm": 3.864333152770996, "learning_rate": 8.93939393939394e-06, "num_tokens": 640362.0, "completions/mean_length": 79.75, "completions/min_length": 17.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 18.0, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 19.0, "rewards/meter/mean": 0.35554254055023193, "rewards/meter/std": 0.4149716794490814, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.6312500238418579, "rewards/judge_quality/std": 0.33417007327079773, "rewards/total_composite/mean": 0.4977003335952759, "rewards/total_composite/std": 0.3186092674732208, "reward": 0.4977003335952759, "reward_std": 0.3186092674732208, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24403917789459229, "sampling/sampling_logp_difference/max": 1.3631038665771484, "sampling/importance_sampling_ratio/min": 0.2558653950691223, "sampling/importance_sampling_ratio/mean": 1.0408650636672974, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.172814294695854, "clip_ratio/low_mean": 0.07537839887663722, "clip_ratio/low_min": 0.07537839887663722, "clip_ratio/high_mean": 0.09027778077870607, "clip_ratio/high_max": 0.09027778077870607, "clip_ratio/region_mean": 0.1656561796553433, "reward_total_mean": 0.4977003335952759, "reward_meter_mean": 0.35554254055023193, "reward_meter_std": 0.4149716794490814, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.6312500238418579, "reward_judge_quality_std": 0.33417007327079773, "reward_total_composite_mean": 0.4977003335952759, "reward_total_composite_std": 0.3186092674732208} {"timestamp_utc": "2026-04-13T08:14:03Z", "mode": "train", "global_step": 352, "epoch": 0.03535911602209945, "loss": 0.0587, "grad_norm": 17.360363006591797, "learning_rate": 8.936363636363638e-06, "num_tokens": 641851.0, "completions/mean_length": 33.125, "completions/min_length": 28.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.125, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.7030478715896606, "rewards/meter/std": 0.40410366654396057, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.989193320274353, "rewards/repeat_soft/std": 0.01741654798388481, "rewards/judge_quality/mean": 0.59375, "rewards/judge_quality/std": 0.22965426743030548, "rewards/total_composite/mean": 0.6477556228637695, "rewards/total_composite/std": 0.22216816246509552, "reward": 0.6477556228637695, "reward_std": 0.22216814756393433, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22128766775131226, "sampling/sampling_logp_difference/max": 1.7048578262329102, "sampling/importance_sampling_ratio/min": 0.18179823458194733, "sampling/importance_sampling_ratio/mean": 1.0677930116653442, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.238470569252968, "clip_ratio/low_mean": 0.11027043638750911, "clip_ratio/low_min": 0.11027043638750911, "clip_ratio/high_mean": 0.051785715855658054, "clip_ratio/high_max": 0.051785715855658054, "clip_ratio/region_mean": 0.16205615224316716, "reward_total_mean": 0.6477556228637695, "reward_meter_mean": 0.7030478715896606, "reward_meter_std": 0.40410366654396057, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.989193320274353, "reward_repeat_soft_std": 0.01741654798388481, "reward_judge_quality_mean": 0.59375, "reward_judge_quality_std": 0.22965426743030548, "reward_total_composite_mean": 0.6477556228637695, "reward_total_composite_std": 0.22216816246509552} {"timestamp_utc": "2026-04-13T08:14:15Z", "mode": "train", "global_step": 353, "epoch": 0.03545956805625314, "loss": 0.161, "grad_norm": 14.564729690551758, "learning_rate": 8.933333333333333e-06, "num_tokens": 643586.0, "completions/mean_length": 37.875, "completions/min_length": 28.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.875, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.5424282550811768, "rewards/meter/std": 0.37805813550949097, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9600545763969421, "rewards/repeat_soft/std": 0.09529281407594681, "rewards/judge_quality/mean": 0.65625, "rewards/judge_quality/std": 0.23790381848812103, "rewards/total_composite/mean": 0.5965635776519775, "rewards/total_composite/std": 0.22408899664878845, "reward": 0.5965635776519775, "reward_std": 0.22408899664878845, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18450625240802765, "sampling/sampling_logp_difference/max": 3.021036148071289, "sampling/importance_sampling_ratio/min": 0.048750679939985275, "sampling/importance_sampling_ratio/mean": 1.0115100145339966, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6726677417755127, "clip_ratio/low_mean": 0.07034111954271793, "clip_ratio/low_min": 0.07034111954271793, "clip_ratio/high_mean": 0.07265911251306534, "clip_ratio/high_max": 0.07265911251306534, "clip_ratio/region_mean": 0.14300023205578327, "reward_total_mean": 0.5965635776519775, "reward_meter_mean": 0.5424282550811768, "reward_meter_std": 0.37805813550949097, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9600545763969421, "reward_repeat_soft_std": 0.09529281407594681, "reward_judge_quality_mean": 0.65625, "reward_judge_quality_std": 0.23790381848812103, "reward_total_composite_mean": 0.5965635776519775, "reward_total_composite_std": 0.22408899664878845} {"timestamp_utc": "2026-04-13T08:14:22Z", "mode": "train", "global_step": 354, "epoch": 0.03556002009040683, "loss": 0.0566, "grad_norm": 19.15772819519043, "learning_rate": 8.930303030303032e-06, "num_tokens": 645169.0, "completions/mean_length": 33.875, "completions/min_length": 27.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.875, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.48136985301971436, "rewards/meter/std": 0.3475463092327118, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9976884126663208, "rewards/repeat_soft/std": 0.002912714844569564, "rewards/judge_quality/mean": 0.7950000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.582542359828949, "rewards/total_composite/std": 0.1957775503396988, "reward": 0.582542359828949, "reward_std": 0.1957775354385376, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17875388264656067, "sampling/sampling_logp_difference/max": 1.5533866882324219, "sampling/importance_sampling_ratio/min": 0.21153037250041962, "sampling/importance_sampling_ratio/mean": 1.0230615139007568, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3181058838963509, "clip_ratio/low_mean": 0.09575557895004749, "clip_ratio/low_min": 0.09575557895004749, "clip_ratio/high_mean": 0.05081847682595253, "clip_ratio/high_max": 0.05081847682595253, "clip_ratio/region_mean": 0.14657405577600002, "reward_total_mean": 0.582542359828949, "reward_meter_mean": 0.48136985301971436, "reward_meter_std": 0.3475463092327118, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9976884126663208, "reward_repeat_soft_std": 0.002912714844569564, "reward_judge_quality_mean": 0.7950000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.582542359828949, "reward_total_composite_std": 0.1957775503396988} {"timestamp_utc": "2026-04-13T08:14:35Z", "mode": "train", "global_step": 355, "epoch": 0.03566047212456052, "loss": 0.0768, "grad_norm": 12.950587272644043, "learning_rate": 8.927272727272728e-06, "num_tokens": 646596.0, "completions/mean_length": 36.375, "completions/min_length": 31.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.375, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.27759212255477905, "rewards/meter/std": 0.1785835325717926, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9925885200500488, "rewards/repeat_soft/std": 0.007694505620747805, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.44296160340309143, "rewards/total_composite/std": 0.08418701589107513, "reward": 0.44296160340309143, "reward_std": 0.08418700098991394, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22627484798431396, "sampling/sampling_logp_difference/max": 2.520127296447754, "sampling/importance_sampling_ratio/min": 0.08044936507940292, "sampling/importance_sampling_ratio/mean": 1.015830159187317, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.899744674563408, "clip_ratio/low_mean": 0.08606674242764711, "clip_ratio/low_min": 0.08606674242764711, "clip_ratio/high_mean": 0.14343082159757614, "clip_ratio/high_max": 0.14343082159757614, "clip_ratio/region_mean": 0.22949756402522326, "reward_total_mean": 0.44296160340309143, "reward_meter_mean": 0.27759212255477905, "reward_meter_std": 0.1785835325717926, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9925885200500488, "reward_repeat_soft_std": 0.007694505620747805, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.44296160340309143, "reward_total_composite_std": 0.08418701589107513} {"timestamp_utc": "2026-04-13T08:14:54Z", "mode": "train", "global_step": 356, "epoch": 0.035760924158714215, "loss": -0.1054, "grad_norm": 3.625981330871582, "learning_rate": 8.924242424242425e-06, "num_tokens": 648292.0, "completions/mean_length": 182.0, "completions/min_length": 38.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 72.0, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.43331071734428406, "rewards/meter/std": 0.39519160985946655, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.18898223340511322, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.9938517808914185, "rewards/repeat_soft/std": 0.008353594690561295, "rewards/judge_quality/mean": 0.22999998927116394, "rewards/judge_quality/std": 0.1437259316444397, "rewards/total_composite/mean": 0.2760443389415741, "rewards/total_composite/std": 0.23782400786876678, "reward": 0.2760443389415741, "reward_std": 0.23782400786876678, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24254828691482544, "sampling/sampling_logp_difference/max": 1.1382880210876465, "sampling/importance_sampling_ratio/min": 0.3203670382499695, "sampling/importance_sampling_ratio/mean": 1.0883963108062744, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.182831972837448, "clip_ratio/low_mean": 0.024671053513884544, "clip_ratio/low_min": 0.024671053513884544, "clip_ratio/high_mean": 0.12995248660445213, "clip_ratio/high_max": 0.12995248660445213, "clip_ratio/region_mean": 0.15462354011833668, "reward_total_mean": 0.2760443389415741, "reward_meter_mean": 0.43331071734428406, "reward_meter_std": 0.39519160985946655, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.18898223340511322, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.9938517808914185, "reward_repeat_soft_std": 0.008353594690561295, "reward_judge_quality_mean": 0.22999998927116394, "reward_judge_quality_std": 0.1437259316444397, "reward_total_composite_mean": 0.2760443389415741, "reward_total_composite_std": 0.23782400786876678} {"timestamp_utc": "2026-04-13T08:15:10Z", "mode": "train", "global_step": 357, "epoch": 0.0358613761928679, "loss": 0.0534, "grad_norm": 4.924477577209473, "learning_rate": 8.921212121212122e-06, "num_tokens": 650441.0, "completions/mean_length": 151.625, "completions/min_length": 42.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 100.14286041259766, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 214.0, "rewards/meter/mean": 0.8353273868560791, "rewards/meter/std": 0.1646488457918167, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2121320366859436, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.8883692622184753, "rewards/repeat_soft/std": 0.17075568437576294, "rewards/judge_quality/mean": 0.2724999785423279, "rewards/judge_quality/std": 0.14119388163089752, "rewards/total_composite/mean": 0.353649377822876, "rewards/total_composite/std": 0.23280341923236847, "reward": 0.353649377822876, "reward_std": 0.23280341923236847, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22231870889663696, "sampling/sampling_logp_difference/max": 1.5295047760009766, "sampling/importance_sampling_ratio/min": 0.21664293110370636, "sampling/importance_sampling_ratio/mean": 1.046348214149475, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.9584960341453552, "clip_ratio/low_mean": 0.03665999136865139, "clip_ratio/low_min": 0.03665999136865139, "clip_ratio/high_mean": 0.11384666338562965, "clip_ratio/high_max": 0.11384666338562965, "clip_ratio/region_mean": 0.15050665475428104, "reward_total_mean": 0.353649377822876, "reward_meter_mean": 0.8353273868560791, "reward_meter_std": 0.1646488457918167, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2121320366859436, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.8883692622184753, "reward_repeat_soft_std": 0.17075568437576294, "reward_judge_quality_mean": 0.2724999785423279, "reward_judge_quality_std": 0.14119388163089752, "reward_total_composite_mean": 0.353649377822876, "reward_total_composite_std": 0.23280341923236847} {"timestamp_utc": "2026-04-13T08:15:18Z", "mode": "train", "global_step": 358, "epoch": 0.0359618282270216, "loss": 0.1867, "grad_norm": 12.765982627868652, "learning_rate": 8.91818181818182e-06, "num_tokens": 652183.0, "completions/mean_length": 59.75, "completions/min_length": 41.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.75, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.5446306467056274, "rewards/meter/std": 0.4207291305065155, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9915409088134766, "rewards/repeat_soft/std": 0.010725677944719791, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.47466665506362915, "rewards/total_composite/std": 0.1031181737780571, "reward": 0.47466665506362915, "reward_std": 0.1031181663274765, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2030535191297531, "sampling/sampling_logp_difference/max": 1.5430059432983398, "sampling/importance_sampling_ratio/min": 0.21373766660690308, "sampling/importance_sampling_ratio/mean": 1.0571295022964478, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6188998222351074, "clip_ratio/low_mean": 0.0892687551677227, "clip_ratio/low_min": 0.0892687551677227, "clip_ratio/high_mean": 0.07147535216063261, "clip_ratio/high_max": 0.07147535216063261, "clip_ratio/region_mean": 0.1607441073283553, "reward_total_mean": 0.47466665506362915, "reward_meter_mean": 0.5446306467056274, "reward_meter_std": 0.4207291305065155, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9915409088134766, "reward_repeat_soft_std": 0.010725677944719791, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.47466665506362915, "reward_total_composite_std": 0.1031181737780571} {"timestamp_utc": "2026-04-13T08:15:28Z", "mode": "train", "global_step": 359, "epoch": 0.03606228026117529, "loss": 0.5258, "grad_norm": 9.785633087158203, "learning_rate": 8.915151515151515e-06, "num_tokens": 654457.0, "completions/mean_length": 101.25, "completions/min_length": 66.0, "completions/max_length": 275.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 101.25, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 275.0, "rewards/meter/mean": 0.6811692714691162, "rewards/meter/std": 0.37058714032173157, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9920492172241211, "rewards/repeat_soft/std": 0.01068015955388546, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.22211645543575287, "rewards/total_composite/mean": 0.5265733003616333, "rewards/total_composite/std": 0.2566831409931183, "reward": 0.5265733003616333, "reward_std": 0.2566831111907959, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22395886480808258, "sampling/sampling_logp_difference/max": 2.0679333209991455, "sampling/importance_sampling_ratio/min": 0.12644682824611664, "sampling/importance_sampling_ratio/mean": 1.0653128623962402, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.974060446023941, "clip_ratio/low_mean": 0.05053030326962471, "clip_ratio/low_min": 0.05053030326962471, "clip_ratio/high_mean": 0.13463346101343632, "clip_ratio/high_max": 0.13463346101343632, "clip_ratio/region_mean": 0.18516376428306103, "reward_total_mean": 0.5265733003616333, "reward_meter_mean": 0.6811692714691162, "reward_meter_std": 0.37058714032173157, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9920492172241211, "reward_repeat_soft_std": 0.01068015955388546, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.22211645543575287, "reward_total_composite_mean": 0.5265733003616333, "reward_total_composite_std": 0.2566831409931183} {"timestamp_utc": "2026-04-13T08:15:38Z", "mode": "train", "global_step": 360, "epoch": 0.03616273229532898, "loss": 0.5822, "grad_norm": 9.903790473937988, "learning_rate": 8.912121212121214e-06, "num_tokens": 656499.0, "completions/mean_length": 91.25, "completions/min_length": 46.0, "completions/max_length": 340.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.25, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 340.0, "rewards/meter/mean": 0.3474387526512146, "rewards/meter/std": 0.32828089594841003, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.15430334210395813, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9982553720474243, "rewards/repeat_soft/std": 0.0026103267446160316, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.13845446705818176, "rewards/total_composite/mean": 0.32856714725494385, "rewards/total_composite/std": 0.2217308133840561, "reward": 0.32856714725494385, "reward_std": 0.2217308133840561, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20932020246982574, "sampling/sampling_logp_difference/max": 1.7218084335327148, "sampling/importance_sampling_ratio/min": 0.17874261736869812, "sampling/importance_sampling_ratio/mean": 1.0371640920639038, "sampling/importance_sampling_ratio/max": 1.958345651626587, "entropy": 4.028614193201065, "clip_ratio/low_mean": 0.022341629955917597, "clip_ratio/low_min": 0.022341629955917597, "clip_ratio/high_mean": 0.14197558723390102, "clip_ratio/high_max": 0.14197558723390102, "clip_ratio/region_mean": 0.16431721718981862, "reward_total_mean": 0.32856714725494385, "reward_meter_mean": 0.3474387526512146, "reward_meter_std": 0.32828089594841003, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.15430334210395813, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9982553720474243, "reward_repeat_soft_std": 0.0026103267446160316, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.13845446705818176, "reward_total_composite_mean": 0.32856714725494385, "reward_total_composite_std": 0.2217308133840561} {"timestamp_utc": "2026-04-13T08:15:50Z", "mode": "train", "global_step": 361, "epoch": 0.036263184329482674, "loss": -0.1497, "grad_norm": 3.4068005084991455, "learning_rate": 8.90909090909091e-06, "num_tokens": 658196.0, "completions/mean_length": 113.125, "completions/min_length": 51.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 56.142860412597656, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.7484950423240662, "rewards/meter/std": 0.31213483214378357, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9944742918014526, "rewards/repeat_soft/std": 0.007948680780827999, "rewards/judge_quality/mean": 0.29750001430511475, "rewards/judge_quality/std": 0.14518460631370544, "rewards/total_composite/mean": 0.44896334409713745, "rewards/total_composite/std": 0.20162814855575562, "reward": 0.44896334409713745, "reward_std": 0.2016281634569168, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.25105538964271545, "sampling/sampling_logp_difference/max": 2.721737861633301, "sampling/importance_sampling_ratio/min": 0.06576037406921387, "sampling/importance_sampling_ratio/mean": 1.0676491260528564, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.2537022531032562, "clip_ratio/low_mean": 0.04337772913277149, "clip_ratio/low_min": 0.04337772913277149, "clip_ratio/high_mean": 0.12420917861163616, "clip_ratio/high_max": 0.12420917861163616, "clip_ratio/region_mean": 0.16758690774440765, "reward_total_mean": 0.44896334409713745, "reward_meter_mean": 0.7484950423240662, "reward_meter_std": 0.31213483214378357, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9944742918014526, "reward_repeat_soft_std": 0.007948680780827999, "reward_judge_quality_mean": 0.29750001430511475, "reward_judge_quality_std": 0.14518460631370544, "reward_total_composite_mean": 0.44896334409713745, "reward_total_composite_std": 0.20162814855575562} {"timestamp_utc": "2026-04-13T08:15:56Z", "mode": "train", "global_step": 362, "epoch": 0.03636363636363636, "loss": -0.0354, "grad_norm": 19.476285934448242, "learning_rate": 8.906060606060607e-06, "num_tokens": 659666.0, "completions/mean_length": 22.75, "completions/min_length": 19.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.75, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.6831570863723755, "rewards/meter/std": 0.31254181265830994, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9976022839546204, "rewards/repeat_soft/std": 0.003977146465331316, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.0645727664232254, "rewards/total_composite/mean": 0.539945125579834, "rewards/total_composite/std": 0.0878436267375946, "reward": 0.539945125579834, "reward_std": 0.08784361928701401, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2205607146024704, "sampling/sampling_logp_difference/max": 1.830787181854248, "sampling/importance_sampling_ratio/min": 0.16028733551502228, "sampling/importance_sampling_ratio/mean": 1.0382581949234009, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4231555983424187, "clip_ratio/low_mean": 0.06328320689499378, "clip_ratio/low_min": 0.06328320689499378, "clip_ratio/high_mean": 0.1684735119342804, "clip_ratio/high_max": 0.1684735119342804, "clip_ratio/region_mean": 0.23175671882927418, "reward_total_mean": 0.539945125579834, "reward_meter_mean": 0.6831570863723755, "reward_meter_std": 0.31254181265830994, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9976022839546204, "reward_repeat_soft_std": 0.003977146465331316, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.0645727664232254, "reward_total_composite_mean": 0.539945125579834, "reward_total_composite_std": 0.0878436267375946} {"timestamp_utc": "2026-04-13T08:16:08Z", "mode": "train", "global_step": 363, "epoch": 0.036464088397790057, "loss": -0.1141, "grad_norm": 4.718713760375977, "learning_rate": 8.903030303030304e-06, "num_tokens": 661668.0, "completions/mean_length": 132.25, "completions/min_length": 55.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 78.0, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.3226236402988434, "rewards/meter/std": 0.3042055368423462, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.987562894821167, "rewards/repeat_soft/std": 0.01699833758175373, "rewards/judge_quality/mean": 0.5400000214576721, "rewards/judge_quality/std": 0.25427767634391785, "rewards/total_composite/mean": 0.4441455006599426, "rewards/total_composite/std": 0.2318834662437439, "reward": 0.4441455006599426, "reward_std": 0.2318834513425827, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21444430947303772, "sampling/sampling_logp_difference/max": 1.1028151512145996, "sampling/importance_sampling_ratio/min": 0.331935316324234, "sampling/importance_sampling_ratio/mean": 1.0466419458389282, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.461810067296028, "clip_ratio/low_mean": 0.085496561601758, "clip_ratio/low_min": 0.085496561601758, "clip_ratio/high_mean": 0.06669638678431511, "clip_ratio/high_max": 0.06669638678431511, "clip_ratio/region_mean": 0.1521929483860731, "reward_total_mean": 0.4441455006599426, "reward_meter_mean": 0.3226236402988434, "reward_meter_std": 0.3042055368423462, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.987562894821167, "reward_repeat_soft_std": 0.01699833758175373, "reward_judge_quality_mean": 0.5400000214576721, "reward_judge_quality_std": 0.25427767634391785, "reward_total_composite_mean": 0.4441455006599426, "reward_total_composite_std": 0.2318834662437439} {"timestamp_utc": "2026-04-13T08:16:15Z", "mode": "train", "global_step": 364, "epoch": 0.036564540431943744, "loss": 0.0201, "grad_norm": 17.85727882385254, "learning_rate": 8.900000000000001e-06, "num_tokens": 663105.0, "completions/mean_length": 33.625, "completions/min_length": 30.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.625, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.8145196437835693, "rewards/meter/std": 0.3323499262332916, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9631528854370117, "rewards/repeat_soft/std": 0.031031915917992592, "rewards/judge_quality/mean": 0.7275000214576721, "rewards/judge_quality/std": 0.2188117355108261, "rewards/total_composite/mean": 0.7170701026916504, "rewards/total_composite/std": 0.19379006326198578, "reward": 0.7170701026916504, "reward_std": 0.19379007816314697, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15746693313121796, "sampling/sampling_logp_difference/max": 1.4978742599487305, "sampling/importance_sampling_ratio/min": 0.2236049920320511, "sampling/importance_sampling_ratio/mean": 1.0556191205978394, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5243082791566849, "clip_ratio/low_mean": 0.04288856312632561, "clip_ratio/low_min": 0.04288856312632561, "clip_ratio/high_mean": 0.09164384286850691, "clip_ratio/high_max": 0.09164384286850691, "clip_ratio/region_mean": 0.13453240599483252, "reward_total_mean": 0.7170701026916504, "reward_meter_mean": 0.8145196437835693, "reward_meter_std": 0.3323499262332916, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9631528854370117, "reward_repeat_soft_std": 0.031031915917992592, "reward_judge_quality_mean": 0.7275000214576721, "reward_judge_quality_std": 0.2188117355108261, "reward_total_composite_mean": 0.7170701026916504, "reward_total_composite_std": 0.19379006326198578} {"timestamp_utc": "2026-04-13T08:16:23Z", "mode": "train", "global_step": 365, "epoch": 0.03666499246609744, "loss": 0.0139, "grad_norm": 11.384407043457031, "learning_rate": 8.896969696969697e-06, "num_tokens": 665024.0, "completions/mean_length": 63.875, "completions/min_length": 60.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.875, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.5080638527870178, "rewards/meter/std": 0.3114967942237854, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9727797508239746, "rewards/repeat_soft/std": 0.061003249138593674, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.47228801250457764, "rewards/total_composite/std": 0.07512867450714111, "reward": 0.47228801250457764, "reward_std": 0.07512866705656052, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.207632914185524, "sampling/sampling_logp_difference/max": 1.5510435104370117, "sampling/importance_sampling_ratio/min": 0.21202661097049713, "sampling/importance_sampling_ratio/mean": 1.0509876012802124, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.7623439729213715, "clip_ratio/low_mean": 0.09796436131000519, "clip_ratio/low_min": 0.09796436131000519, "clip_ratio/high_mean": 0.06974686309695244, "clip_ratio/high_max": 0.06974686309695244, "clip_ratio/region_mean": 0.16771122440695763, "reward_total_mean": 0.47228801250457764, "reward_meter_mean": 0.5080638527870178, "reward_meter_std": 0.3114967942237854, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9727797508239746, "reward_repeat_soft_std": 0.061003249138593674, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.47228801250457764, "reward_total_composite_std": 0.07512867450714111} {"timestamp_utc": "2026-04-13T08:16:29Z", "mode": "train", "global_step": 366, "epoch": 0.036765444500251133, "loss": 0.0729, "grad_norm": 22.787700653076172, "learning_rate": 8.893939393939394e-06, "num_tokens": 666534.0, "completions/mean_length": 28.75, "completions/min_length": 24.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.75, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.7012413740158081, "rewards/meter/std": 0.42376846075057983, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9953725934028625, "rewards/repeat_soft/std": 0.004975995514541864, "rewards/judge_quality/mean": 0.6112500429153442, "rewards/judge_quality/std": 0.25587037205696106, "rewards/total_composite/mean": 0.6596815586090088, "rewards/total_composite/std": 0.2462015450000763, "reward": 0.6596815586090088, "reward_std": 0.2462015450000763, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.25858786702156067, "sampling/sampling_logp_difference/max": 2.060631275177002, "sampling/importance_sampling_ratio/min": 0.12737353146076202, "sampling/importance_sampling_ratio/mean": 1.077356219291687, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6123074293136597, "clip_ratio/low_mean": 0.13816550932824612, "clip_ratio/low_min": 0.13816550932824612, "clip_ratio/high_mean": 0.053494623862206936, "clip_ratio/high_max": 0.053494623862206936, "clip_ratio/region_mean": 0.19166013319045305, "reward_total_mean": 0.6596815586090088, "reward_meter_mean": 0.7012413740158081, "reward_meter_std": 0.42376846075057983, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9953725934028625, "reward_repeat_soft_std": 0.004975995514541864, "reward_judge_quality_mean": 0.6112500429153442, "reward_judge_quality_std": 0.25587037205696106, "reward_total_composite_mean": 0.6596815586090088, "reward_total_composite_std": 0.2462015450000763} {"timestamp_utc": "2026-04-13T08:16:41Z", "mode": "train", "global_step": 367, "epoch": 0.03686589653440482, "loss": -0.1425, "grad_norm": 2.678206205368042, "learning_rate": 8.890909090909091e-06, "num_tokens": 668583.0, "completions/mean_length": 206.125, "completions/min_length": 77.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 104.16667175292969, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 156.0, "rewards/meter/mean": 0.27708929777145386, "rewards/meter/std": 0.27479180693626404, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.25877460837364197, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9985931515693665, "rewards/repeat_soft/std": 0.0012916075065732002, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.22238561511039734, "rewards/total_composite/mean": 0.3461523652076721, "rewards/total_composite/std": 0.24402405321598053, "reward": 0.3461523652076721, "reward_std": 0.24402403831481934, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22072133421897888, "sampling/sampling_logp_difference/max": 1.3632798194885254, "sampling/importance_sampling_ratio/min": 0.2558203637599945, "sampling/importance_sampling_ratio/mean": 1.029079556465149, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9468597173690796, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.15607979893684387, "clip_ratio/high_max": 0.15607979893684387, "clip_ratio/region_mean": 0.15607979893684387, "reward_total_mean": 0.3461523652076721, "reward_meter_mean": 0.27708929777145386, "reward_meter_std": 0.27479180693626404, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.25877460837364197, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9985931515693665, "reward_repeat_soft_std": 0.0012916075065732002, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.22238561511039734, "reward_total_composite_mean": 0.3461523652076721, "reward_total_composite_std": 0.24402405321598053} {"timestamp_utc": "2026-04-13T08:16:57Z", "mode": "train", "global_step": 368, "epoch": 0.036966348568558516, "loss": -0.0744, "grad_norm": 4.494978427886963, "learning_rate": 8.887878787878789e-06, "num_tokens": 670159.0, "completions/mean_length": 96.0, "completions/min_length": 31.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 36.57143020629883, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.5928663015365601, "rewards/meter/std": 0.37849077582359314, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9926130175590515, "rewards/repeat_soft/std": 0.012723951600492, "rewards/judge_quality/mean": 0.5400000214576721, "rewards/judge_quality/std": 0.2958281636238098, "rewards/total_composite/mean": 0.49554625153541565, "rewards/total_composite/std": 0.2634750008583069, "reward": 0.49554625153541565, "reward_std": 0.2634750008583069, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21571017801761627, "sampling/sampling_logp_difference/max": 1.3126220703125, "sampling/importance_sampling_ratio/min": 0.2691135108470917, "sampling/importance_sampling_ratio/mean": 1.060481071472168, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.405510425567627, "clip_ratio/low_mean": 0.057502121198922396, "clip_ratio/low_min": 0.057502121198922396, "clip_ratio/high_mean": 0.08323585148900747, "clip_ratio/high_max": 0.08323585148900747, "clip_ratio/region_mean": 0.14073797268792987, "reward_total_mean": 0.49554625153541565, "reward_meter_mean": 0.5928663015365601, "reward_meter_std": 0.37849077582359314, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9926130175590515, "reward_repeat_soft_std": 0.012723951600492, "reward_judge_quality_mean": 0.5400000214576721, "reward_judge_quality_std": 0.2958281636238098, "reward_total_composite_mean": 0.49554625153541565, "reward_total_composite_std": 0.2634750008583069} {"timestamp_utc": "2026-04-13T08:17:04Z", "mode": "train", "global_step": 369, "epoch": 0.037066800602712204, "loss": 0.0322, "grad_norm": 12.926513671875, "learning_rate": 8.884848484848486e-06, "num_tokens": 671859.0, "completions/mean_length": 55.5, "completions/min_length": 43.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.5, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.3802972435951233, "rewards/meter/std": 0.39531365036964417, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.15430334210395813, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9928926229476929, "rewards/repeat_soft/std": 0.010803640820086002, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.12906256318092346, "rewards/total_composite/mean": 0.45972201228141785, "rewards/total_composite/std": 0.17263850569725037, "reward": 0.45972201228141785, "reward_std": 0.17263852059841156, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22510534524917603, "sampling/sampling_logp_difference/max": 1.3513402938842773, "sampling/importance_sampling_ratio/min": 0.25889304280281067, "sampling/importance_sampling_ratio/mean": 1.0685721635818481, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.410489469766617, "clip_ratio/low_mean": 0.08347007166594267, "clip_ratio/low_min": 0.08347007166594267, "clip_ratio/high_mean": 0.05272738356143236, "clip_ratio/high_max": 0.05272738356143236, "clip_ratio/region_mean": 0.13619745522737503, "reward_total_mean": 0.45972201228141785, "reward_meter_mean": 0.3802972435951233, "reward_meter_std": 0.39531365036964417, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.15430334210395813, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9928926229476929, "reward_repeat_soft_std": 0.010803640820086002, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.12906256318092346, "reward_total_composite_mean": 0.45972201228141785, "reward_total_composite_std": 0.17263850569725037} {"timestamp_utc": "2026-04-13T08:17:11Z", "mode": "train", "global_step": 370, "epoch": 0.0371672526368659, "loss": 0.0327, "grad_norm": 38.586856842041016, "learning_rate": 8.881818181818183e-06, "num_tokens": 673312.0, "completions/mean_length": 23.625, "completions/min_length": 22.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.625, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.2130904495716095, "rewards/meter/std": 0.22513271868228912, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9993371367454529, "rewards/repeat_soft/std": 0.0018749026348814368, "rewards/judge_quality/mean": 0.7649999856948853, "rewards/judge_quality/std": 0.2979933023452759, "rewards/total_composite/mean": 0.4061920940876007, "rewards/total_composite/std": 0.2044784426689148, "reward": 0.4061920940876007, "reward_std": 0.2044784426689148, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19987602531909943, "sampling/sampling_logp_difference/max": 1.7916243076324463, "sampling/importance_sampling_ratio/min": 0.1666891872882843, "sampling/importance_sampling_ratio/mean": 1.0216882228851318, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1096134707331657, "clip_ratio/low_mean": 0.1515974961221218, "clip_ratio/low_min": 0.1515974961221218, "clip_ratio/high_mean": 0.04985007457435131, "clip_ratio/high_max": 0.04985007457435131, "clip_ratio/region_mean": 0.20144757069647312, "reward_total_mean": 0.4061920940876007, "reward_meter_mean": 0.2130904495716095, "reward_meter_std": 0.22513271868228912, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9993371367454529, "reward_repeat_soft_std": 0.0018749026348814368, "reward_judge_quality_mean": 0.7649999856948853, "reward_judge_quality_std": 0.2979933023452759, "reward_total_composite_mean": 0.4061920940876007, "reward_total_composite_std": 0.2044784426689148} {"timestamp_utc": "2026-04-13T08:17:18Z", "mode": "train", "global_step": 371, "epoch": 0.037267704671019586, "loss": 0.0258, "grad_norm": 17.992992401123047, "learning_rate": 8.87878787878788e-06, "num_tokens": 674861.0, "completions/mean_length": 31.625, "completions/min_length": 28.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.625, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.815354585647583, "rewards/meter/std": 0.30784371495246887, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9822838306427002, "rewards/repeat_soft/std": 0.011374717578291893, "rewards/judge_quality/mean": 0.6312500238418579, "rewards/judge_quality/std": 0.22158117592334747, "rewards/total_composite/mean": 0.6992222666740417, "rewards/total_composite/std": 0.1985442191362381, "reward": 0.6992222666740417, "reward_std": 0.1985442191362381, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18011300265789032, "sampling/sampling_logp_difference/max": 1.3447229862213135, "sampling/importance_sampling_ratio/min": 0.260611891746521, "sampling/importance_sampling_ratio/mean": 1.0152212381362915, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.919819250702858, "clip_ratio/low_mean": 0.06414490658789873, "clip_ratio/low_min": 0.06414490658789873, "clip_ratio/high_mean": 0.06656280532479286, "clip_ratio/high_max": 0.06656280532479286, "clip_ratio/region_mean": 0.1307077119126916, "reward_total_mean": 0.6992222666740417, "reward_meter_mean": 0.815354585647583, "reward_meter_std": 0.30784371495246887, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9822838306427002, "reward_repeat_soft_std": 0.011374717578291893, "reward_judge_quality_mean": 0.6312500238418579, "reward_judge_quality_std": 0.22158117592334747, "reward_total_composite_mean": 0.6992222666740417, "reward_total_composite_std": 0.1985442191362381} {"timestamp_utc": "2026-04-13T08:17:25Z", "mode": "train", "global_step": 372, "epoch": 0.03736815670517328, "loss": 0.0014, "grad_norm": 16.33405303955078, "learning_rate": 8.875757575757576e-06, "num_tokens": 676335.0, "completions/mean_length": 21.25, "completions/min_length": 16.0, "completions/max_length": 24.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.25, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 24.0, "rewards/meter/mean": 0.16977465152740479, "rewards/meter/std": 0.28688836097717285, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9377456903457642, "rewards/repeat_soft/std": 0.032807815819978714, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.3870103359222412, "rewards/total_composite/std": 0.07520600408315659, "reward": 0.3870103359222412, "reward_std": 0.07520599663257599, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17178981006145477, "sampling/sampling_logp_difference/max": 1.4719525575637817, "sampling/importance_sampling_ratio/min": 0.22947698831558228, "sampling/importance_sampling_ratio/mean": 1.0394285917282104, "sampling/importance_sampling_ratio/max": 1.8622790575027466, "entropy": 1.3809120282530785, "clip_ratio/low_mean": 0.09722041804343462, "clip_ratio/low_min": 0.09722041804343462, "clip_ratio/high_mean": 0.01733954483643174, "clip_ratio/high_max": 0.01733954483643174, "clip_ratio/region_mean": 0.11455996287986636, "reward_total_mean": 0.3870103359222412, "reward_meter_mean": 0.16977465152740479, "reward_meter_std": 0.28688836097717285, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9377456903457642, "reward_repeat_soft_std": 0.032807815819978714, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.3870103359222412, "reward_total_composite_std": 0.07520600408315659} {"timestamp_utc": "2026-04-13T08:17:34Z", "mode": "train", "global_step": 373, "epoch": 0.03746860873932697, "loss": 0.5053, "grad_norm": 17.594009399414062, "learning_rate": 8.872727272727275e-06, "num_tokens": 678000.0, "completions/mean_length": 46.125, "completions/min_length": 29.0, "completions/max_length": 124.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.125, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.49487289786338806, "rewards/meter/std": 0.493295282125473, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9968066215515137, "rewards/repeat_soft/std": 0.004193334840238094, "rewards/judge_quality/mean": 0.2749999761581421, "rewards/judge_quality/std": 0.155838742852211, "rewards/total_composite/mean": 0.41905075311660767, "rewards/total_composite/std": 0.2089303880929947, "reward": 0.41905075311660767, "reward_std": 0.2089303880929947, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2209448367357254, "sampling/sampling_logp_difference/max": 1.3029823303222656, "sampling/importance_sampling_ratio/min": 0.2717202305793762, "sampling/importance_sampling_ratio/mean": 1.0277589559555054, "sampling/importance_sampling_ratio/max": 1.9530292749404907, "entropy": 4.145709306001663, "clip_ratio/low_mean": 0.0727820647880435, "clip_ratio/low_min": 0.0727820647880435, "clip_ratio/high_mean": 0.1434063334017992, "clip_ratio/high_max": 0.1434063334017992, "clip_ratio/region_mean": 0.2161883981898427, "reward_total_mean": 0.41905075311660767, "reward_meter_mean": 0.49487289786338806, "reward_meter_std": 0.493295282125473, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9968066215515137, "reward_repeat_soft_std": 0.004193334840238094, "reward_judge_quality_mean": 0.2749999761581421, "reward_judge_quality_std": 0.155838742852211, "reward_total_composite_mean": 0.41905075311660767, "reward_total_composite_std": 0.2089303880929947} {"timestamp_utc": "2026-04-13T08:17:40Z", "mode": "train", "global_step": 374, "epoch": 0.03756906077348066, "loss": 0.0485, "grad_norm": 15.2440767288208, "learning_rate": 8.86969696969697e-06, "num_tokens": 679437.0, "completions/mean_length": 31.625, "completions/min_length": 27.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.625, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.7328658699989319, "rewards/meter/std": 0.37381434440612793, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9985873699188232, "rewards/repeat_soft/std": 0.0030330861918628216, "rewards/judge_quality/mean": 0.4649999737739563, "rewards/judge_quality/std": 0.18423588573932648, "rewards/total_composite/mean": 0.599024772644043, "rewards/total_composite/std": 0.1625562608242035, "reward": 0.599024772644043, "reward_std": 0.1625562459230423, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21124495565891266, "sampling/sampling_logp_difference/max": 1.1573410034179688, "sampling/importance_sampling_ratio/min": 0.3143208622932434, "sampling/importance_sampling_ratio/mean": 1.0387402772903442, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.707137167453766, "clip_ratio/low_mean": 0.0766433198004961, "clip_ratio/low_min": 0.0766433198004961, "clip_ratio/high_mean": 0.10901551321148872, "clip_ratio/high_max": 0.10901551321148872, "clip_ratio/region_mean": 0.18565883301198483, "reward_total_mean": 0.599024772644043, "reward_meter_mean": 0.7328658699989319, "reward_meter_std": 0.37381434440612793, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9985873699188232, "reward_repeat_soft_std": 0.0030330861918628216, "reward_judge_quality_mean": 0.4649999737739563, "reward_judge_quality_std": 0.18423588573932648, "reward_total_composite_mean": 0.599024772644043, "reward_total_composite_std": 0.1625562608242035} {"timestamp_utc": "2026-04-13T08:17:47Z", "mode": "train", "global_step": 375, "epoch": 0.03766951280763436, "loss": 0.0524, "grad_norm": 24.896814346313477, "learning_rate": 8.866666666666668e-06, "num_tokens": 681064.0, "completions/mean_length": 34.375, "completions/min_length": 32.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.375, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.7659133672714233, "rewards/meter/std": 0.3337141275405884, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9985313415527344, "rewards/repeat_soft/std": 0.002027380047366023, "rewards/judge_quality/mean": 0.8324999809265137, "rewards/judge_quality/std": 0.18077217042446136, "rewards/total_composite/mean": 0.7856926321983337, "rewards/total_composite/std": 0.20556603372097015, "reward": 0.7856926321983337, "reward_std": 0.20556604862213135, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1270153820514679, "sampling/sampling_logp_difference/max": 1.3475301265716553, "sampling/importance_sampling_ratio/min": 0.3013465404510498, "sampling/importance_sampling_ratio/mean": 1.014768123626709, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8150487747043371, "clip_ratio/low_mean": 0.08566176518797874, "clip_ratio/low_min": 0.08566176518797874, "clip_ratio/high_mean": 0.06432462367229164, "clip_ratio/high_max": 0.06432462367229164, "clip_ratio/region_mean": 0.14998638886027038, "reward_total_mean": 0.7856926321983337, "reward_meter_mean": 0.7659133672714233, "reward_meter_std": 0.3337141275405884, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9985313415527344, "reward_repeat_soft_std": 0.002027380047366023, "reward_judge_quality_mean": 0.8324999809265137, "reward_judge_quality_std": 0.18077217042446136, "reward_total_composite_mean": 0.7856926321983337, "reward_total_composite_std": 0.20556603372097015} {"timestamp_utc": "2026-04-13T08:17:53Z", "mode": "train", "global_step": 376, "epoch": 0.037769964841788045, "loss": 0.0104, "grad_norm": 21.54682731628418, "learning_rate": 8.863636363636365e-06, "num_tokens": 682490.0, "completions/mean_length": 20.25, "completions/min_length": 18.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 20.25, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.8666355013847351, "rewards/meter/std": 0.23355844616889954, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.3687500059604645, "rewards/judge_quality/std": 0.10802611708641052, "rewards/total_composite/mean": 0.5619431734085083, "rewards/total_composite/std": 0.08701276779174805, "reward": 0.5619431734085083, "reward_std": 0.08701277524232864, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2312190681695938, "sampling/sampling_logp_difference/max": 1.1098909378051758, "sampling/importance_sampling_ratio/min": 0.3295949101448059, "sampling/importance_sampling_ratio/mean": 1.0532430410385132, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.901041753590107, "clip_ratio/low_mean": 0.0575396828353405, "clip_ratio/low_min": 0.0575396828353405, "clip_ratio/high_mean": 0.11208952125161886, "clip_ratio/high_max": 0.11208952125161886, "clip_ratio/region_mean": 0.16962920408695936, "reward_total_mean": 0.5619431734085083, "reward_meter_mean": 0.8666355013847351, "reward_meter_std": 0.23355844616889954, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.3687500059604645, "reward_judge_quality_std": 0.10802611708641052, "reward_total_composite_mean": 0.5619431734085083, "reward_total_composite_std": 0.08701276779174805} {"timestamp_utc": "2026-04-13T08:18:00Z", "mode": "train", "global_step": 377, "epoch": 0.03787041687594174, "loss": 0.0295, "grad_norm": 13.482222557067871, "learning_rate": 8.860606060606062e-06, "num_tokens": 684398.0, "completions/mean_length": 63.5, "completions/min_length": 61.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.5, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.8272577524185181, "rewards/meter/std": 0.24018819630146027, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.994218647480011, "rewards/repeat_soft/std": 0.005557127296924591, "rewards/judge_quality/mean": 0.5149999856948853, "rewards/judge_quality/std": 0.18031719326972961, "rewards/total_composite/mean": 0.6300148963928223, "rewards/total_composite/std": 0.13305138051509857, "reward": 0.6300148963928223, "reward_std": 0.133051335811615, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22164733707904816, "sampling/sampling_logp_difference/max": 1.8124122619628906, "sampling/importance_sampling_ratio/min": 0.1632598340511322, "sampling/importance_sampling_ratio/mean": 1.0644114017486572, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.9146456122398376, "clip_ratio/low_mean": 0.09237921983003616, "clip_ratio/low_min": 0.09237921983003616, "clip_ratio/high_mean": 0.09039351716637611, "clip_ratio/high_max": 0.09039351716637611, "clip_ratio/region_mean": 0.18277273699641228, "reward_total_mean": 0.6300148963928223, "reward_meter_mean": 0.8272577524185181, "reward_meter_std": 0.24018819630146027, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.994218647480011, "reward_repeat_soft_std": 0.005557127296924591, "reward_judge_quality_mean": 0.5149999856948853, "reward_judge_quality_std": 0.18031719326972961, "reward_total_composite_mean": 0.6300148963928223, "reward_total_composite_std": 0.13305138051509857} {"timestamp_utc": "2026-04-13T08:18:07Z", "mode": "train", "global_step": 378, "epoch": 0.03797086891009543, "loss": 0.0172, "grad_norm": 11.616564750671387, "learning_rate": 8.857575757575758e-06, "num_tokens": 686446.0, "completions/mean_length": 67.0, "completions/min_length": 61.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.827986478805542, "rewards/meter/std": 0.24381479620933533, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9585182070732117, "rewards/repeat_soft/std": 0.05319053307175636, "rewards/judge_quality/mean": 0.4737499952316284, "rewards/judge_quality/std": 0.16291432082653046, "rewards/total_composite/mean": 0.5980088114738464, "rewards/total_composite/std": 0.11041803658008575, "reward": 0.5980088114738464, "reward_std": 0.11041803658008575, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21592731773853302, "sampling/sampling_logp_difference/max": 1.395995020866394, "sampling/importance_sampling_ratio/min": 0.2612544894218445, "sampling/importance_sampling_ratio/mean": 1.0480549335479736, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.346274048089981, "clip_ratio/low_mean": 0.07993471156805754, "clip_ratio/low_min": 0.07993471156805754, "clip_ratio/high_mean": 0.07965071126818657, "clip_ratio/high_max": 0.07965071126818657, "clip_ratio/region_mean": 0.1595854228362441, "reward_total_mean": 0.5980088114738464, "reward_meter_mean": 0.827986478805542, "reward_meter_std": 0.24381479620933533, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9585182070732117, "reward_repeat_soft_std": 0.05319053307175636, "reward_judge_quality_mean": 0.4737499952316284, "reward_judge_quality_std": 0.16291432082653046, "reward_total_composite_mean": 0.5980088114738464, "reward_total_composite_std": 0.11041803658008575} {"timestamp_utc": "2026-04-13T08:18:14Z", "mode": "train", "global_step": 379, "epoch": 0.03807132094424912, "loss": 0.0941, "grad_norm": 20.9310359954834, "learning_rate": 8.854545454545455e-06, "num_tokens": 688064.0, "completions/mean_length": 32.25, "completions/min_length": 28.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.25, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.8344127535820007, "rewards/meter/std": 0.31773367524147034, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9943695068359375, "rewards/repeat_soft/std": 0.0056973411701619625, "rewards/judge_quality/mean": 0.518750011920929, "rewards/judge_quality/std": 0.17908000946044922, "rewards/total_composite/mean": 0.5999100804328918, "rewards/total_composite/std": 0.08593261241912842, "reward": 0.5999100804328918, "reward_std": 0.08593261241912842, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20274046063423157, "sampling/sampling_logp_difference/max": 1.5597805976867676, "sampling/importance_sampling_ratio/min": 0.21018217504024506, "sampling/importance_sampling_ratio/mean": 1.037693738937378, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8272703140974045, "clip_ratio/low_mean": 0.027027027681469917, "clip_ratio/low_min": 0.027027027681469917, "clip_ratio/high_mean": 0.17235427629202604, "clip_ratio/high_max": 0.17235427629202604, "clip_ratio/region_mean": 0.19938130397349596, "reward_total_mean": 0.5999100804328918, "reward_meter_mean": 0.8344127535820007, "reward_meter_std": 0.31773367524147034, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9943695068359375, "reward_repeat_soft_std": 0.0056973411701619625, "reward_judge_quality_mean": 0.518750011920929, "reward_judge_quality_std": 0.17908000946044922, "reward_total_composite_mean": 0.5999100804328918, "reward_total_composite_std": 0.08593261241912842} {"timestamp_utc": "2026-04-13T08:18:21Z", "mode": "train", "global_step": 380, "epoch": 0.03817177297840281, "loss": -0.0215, "grad_norm": 14.866976737976074, "learning_rate": 8.851515151515152e-06, "num_tokens": 689762.0, "completions/mean_length": 49.25, "completions/min_length": 45.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.25, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.2547784149646759, "rewards/meter/std": 0.32982468605041504, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9860686659812927, "rewards/repeat_soft/std": 0.012328761629760265, "rewards/judge_quality/mean": 0.5487499833106995, "rewards/judge_quality/std": 0.22937417030334473, "rewards/total_composite/mean": 0.4310709834098816, "rewards/total_composite/std": 0.09128675609827042, "reward": 0.4310709834098816, "reward_std": 0.09128675609827042, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2316330373287201, "sampling/sampling_logp_difference/max": 3.0058579444885254, "sampling/importance_sampling_ratio/min": 0.04949627071619034, "sampling/importance_sampling_ratio/mean": 1.0419652462005615, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.449590340256691, "clip_ratio/low_mean": 0.11093355249613523, "clip_ratio/low_min": 0.11093355249613523, "clip_ratio/high_mean": 0.07752490043640137, "clip_ratio/high_max": 0.07752490043640137, "clip_ratio/region_mean": 0.1884584529325366, "reward_total_mean": 0.4310709834098816, "reward_meter_mean": 0.2547784149646759, "reward_meter_std": 0.32982468605041504, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9860686659812927, "reward_repeat_soft_std": 0.012328761629760265, "reward_judge_quality_mean": 0.5487499833106995, "reward_judge_quality_std": 0.22937417030334473, "reward_total_composite_mean": 0.4310709834098816, "reward_total_composite_std": 0.09128675609827042} {"timestamp_utc": "2026-04-13T08:18:28Z", "mode": "train", "global_step": 381, "epoch": 0.038272225012556504, "loss": 0.0262, "grad_norm": 22.278461456298828, "learning_rate": 8.84848484848485e-06, "num_tokens": 691310.0, "completions/mean_length": 33.5, "completions/min_length": 26.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.5, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.6267951130867004, "rewards/meter/std": 0.3689000904560089, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9895439147949219, "rewards/repeat_soft/std": 0.009762046858668327, "rewards/judge_quality/mean": 0.6487500071525574, "rewards/judge_quality/std": 0.2456151396036148, "rewards/total_composite/mean": 0.5943039059638977, "rewards/total_composite/std": 0.31039348244667053, "reward": 0.5943039059638977, "reward_std": 0.31039348244667053, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21351630985736847, "sampling/sampling_logp_difference/max": 2.1638095378875732, "sampling/importance_sampling_ratio/min": 0.11488661915063858, "sampling/importance_sampling_ratio/mean": 1.0162560939788818, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.262278027832508, "clip_ratio/low_mean": 0.09266827069222927, "clip_ratio/low_min": 0.09266827069222927, "clip_ratio/high_mean": 0.08275891002267599, "clip_ratio/high_max": 0.08275891002267599, "clip_ratio/region_mean": 0.17542718071490526, "reward_total_mean": 0.5943039059638977, "reward_meter_mean": 0.6267951130867004, "reward_meter_std": 0.3689000904560089, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9895439147949219, "reward_repeat_soft_std": 0.009762046858668327, "reward_judge_quality_mean": 0.6487500071525574, "reward_judge_quality_std": 0.2456151396036148, "reward_total_composite_mean": 0.5943039059638977, "reward_total_composite_std": 0.31039348244667053} {"timestamp_utc": "2026-04-13T08:18:35Z", "mode": "train", "global_step": 382, "epoch": 0.0383726770467102, "loss": 0.0568, "grad_norm": 13.969280242919922, "learning_rate": 8.845454545454547e-06, "num_tokens": 693068.0, "completions/mean_length": 58.75, "completions/min_length": 53.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.75, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.4683726131916046, "rewards/meter/std": 0.27449503540992737, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.994033694267273, "rewards/repeat_soft/std": 0.00757418991997838, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.4857119619846344, "rewards/total_composite/std": 0.09801746904850006, "reward": 0.4857119619846344, "reward_std": 0.09801746904850006, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21199508011341095, "sampling/sampling_logp_difference/max": 1.337010383605957, "sampling/importance_sampling_ratio/min": 0.2626296579837799, "sampling/importance_sampling_ratio/mean": 1.0761182308197021, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.5622952729463577, "clip_ratio/low_mean": 0.12251085415482521, "clip_ratio/low_min": 0.12251085415482521, "clip_ratio/high_mean": 0.0687596295028925, "clip_ratio/high_max": 0.0687596295028925, "clip_ratio/region_mean": 0.1912704836577177, "reward_total_mean": 0.4857119619846344, "reward_meter_mean": 0.4683726131916046, "reward_meter_std": 0.27449503540992737, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.994033694267273, "reward_repeat_soft_std": 0.00757418991997838, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.4857119619846344, "reward_total_composite_std": 0.09801746904850006} {"timestamp_utc": "2026-04-13T08:18:42Z", "mode": "train", "global_step": 383, "epoch": 0.03847312908086389, "loss": 0.0829, "grad_norm": 14.486379623413086, "learning_rate": 8.842424242424244e-06, "num_tokens": 695375.0, "completions/mean_length": 85.375, "completions/min_length": 78.0, "completions/max_length": 95.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.375, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.41754233837127686, "rewards/meter/std": 0.2040727734565735, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9971754550933838, "rewards/repeat_soft/std": 0.0023173748049885035, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.39841270446777344, "rewards/total_composite/std": 0.17113541066646576, "reward": 0.39841270446777344, "reward_std": 0.17113539576530457, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24437576532363892, "sampling/sampling_logp_difference/max": 1.599085807800293, "sampling/importance_sampling_ratio/min": 0.20208117365837097, "sampling/importance_sampling_ratio/mean": 1.0334339141845703, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.323674261569977, "clip_ratio/low_mean": 0.051868095993995667, "clip_ratio/low_min": 0.051868095993995667, "clip_ratio/high_mean": 0.17507386952638626, "clip_ratio/high_max": 0.17507386952638626, "clip_ratio/region_mean": 0.22694196552038193, "reward_total_mean": 0.39841270446777344, "reward_meter_mean": 0.41754233837127686, "reward_meter_std": 0.2040727734565735, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9971754550933838, "reward_repeat_soft_std": 0.0023173748049885035, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.39841270446777344, "reward_total_composite_std": 0.17113541066646576} {"timestamp_utc": "2026-04-13T08:18:50Z", "mode": "train", "global_step": 384, "epoch": 0.03857358111501758, "loss": -0.0105, "grad_norm": 17.235023498535156, "learning_rate": 8.83939393939394e-06, "num_tokens": 696932.0, "completions/mean_length": 35.625, "completions/min_length": 32.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.625, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.8582241535186768, "rewards/meter/std": 0.31072792410850525, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.995353102684021, "rewards/repeat_soft/std": 0.0044983429834246635, "rewards/judge_quality/mean": 0.5399999618530273, "rewards/judge_quality/std": 0.2218751311302185, "rewards/total_composite/mean": 0.658287763595581, "rewards/total_composite/std": 0.1771986037492752, "reward": 0.658287763595581, "reward_std": 0.1771986037492752, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22254599630832672, "sampling/sampling_logp_difference/max": 1.4940357208251953, "sampling/importance_sampling_ratio/min": 0.22446495294570923, "sampling/importance_sampling_ratio/mean": 1.022464632987976, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8612878322601318, "clip_ratio/low_mean": 0.10957883857190609, "clip_ratio/low_min": 0.10957883857190609, "clip_ratio/high_mean": 0.06023901887238026, "clip_ratio/high_max": 0.06023901887238026, "clip_ratio/region_mean": 0.16981785744428635, "reward_total_mean": 0.658287763595581, "reward_meter_mean": 0.8582241535186768, "reward_meter_std": 0.31072792410850525, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.995353102684021, "reward_repeat_soft_std": 0.0044983429834246635, "reward_judge_quality_mean": 0.5399999618530273, "reward_judge_quality_std": 0.2218751311302185, "reward_total_composite_mean": 0.658287763595581, "reward_total_composite_std": 0.1771986037492752} {"timestamp_utc": "2026-04-13T08:18:56Z", "mode": "train", "global_step": 385, "epoch": 0.03867403314917127, "loss": -0.0119, "grad_norm": 19.83598518371582, "learning_rate": 8.836363636363637e-06, "num_tokens": 698256.0, "completions/mean_length": 18.5, "completions/min_length": 15.0, "completions/max_length": 21.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 18.5, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 21.0, "rewards/meter/mean": 0.5626965761184692, "rewards/meter/std": 0.42163652181625366, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.5029791593551636, "rewards/total_composite/std": 0.12039375305175781, "reward": 0.5029791593551636, "reward_std": 0.12039375305175781, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13141751289367676, "sampling/sampling_logp_difference/max": 1.0829339027404785, "sampling/importance_sampling_ratio/min": 0.33860063552856445, "sampling/importance_sampling_ratio/mean": 1.0016568899154663, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6041314825415611, "clip_ratio/low_mean": 0.050520834513008595, "clip_ratio/low_min": 0.050520834513008595, "clip_ratio/high_mean": 0.05680868960916996, "clip_ratio/high_max": 0.05680868960916996, "clip_ratio/region_mean": 0.10732952412217855, "reward_total_mean": 0.5029791593551636, "reward_meter_mean": 0.5626965761184692, "reward_meter_std": 0.42163652181625366, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.5029791593551636, "reward_total_composite_std": 0.12039375305175781} {"timestamp_utc": "2026-04-13T08:19:03Z", "mode": "train", "global_step": 386, "epoch": 0.038774485183324964, "loss": -0.0405, "grad_norm": 16.2897891998291, "learning_rate": 8.833333333333334e-06, "num_tokens": 700003.0, "completions/mean_length": 46.375, "completions/min_length": 42.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.375, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.6400235891342163, "rewards/meter/std": 0.370656281709671, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9928402900695801, "rewards/repeat_soft/std": 0.011342632584273815, "rewards/judge_quality/mean": 0.6150000095367432, "rewards/judge_quality/std": 0.1538088172674179, "rewards/total_composite/mean": 0.5819841027259827, "rewards/total_composite/std": 0.1263429969549179, "reward": 0.5819841027259827, "reward_std": 0.1263429820537567, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2091495394706726, "sampling/sampling_logp_difference/max": 2.2717998027801514, "sampling/importance_sampling_ratio/min": 0.1031264066696167, "sampling/importance_sampling_ratio/mean": 1.0420584678649902, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9817766696214676, "clip_ratio/low_mean": 0.06466450355947018, "clip_ratio/low_min": 0.06466450355947018, "clip_ratio/high_mean": 0.1348732877522707, "clip_ratio/high_max": 0.1348732877522707, "clip_ratio/region_mean": 0.19953779131174088, "reward_total_mean": 0.5819841027259827, "reward_meter_mean": 0.6400235891342163, "reward_meter_std": 0.370656281709671, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9928402900695801, "reward_repeat_soft_std": 0.011342632584273815, "reward_judge_quality_mean": 0.6150000095367432, "reward_judge_quality_std": 0.1538088172674179, "reward_total_composite_mean": 0.5819841027259827, "reward_total_composite_std": 0.1263429969549179} {"timestamp_utc": "2026-04-13T08:19:09Z", "mode": "train", "global_step": 387, "epoch": 0.03887493721747865, "loss": 0.0497, "grad_norm": 12.300729751586914, "learning_rate": 8.830303030303031e-06, "num_tokens": 701732.0, "completions/mean_length": 50.125, "completions/min_length": 44.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.125, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.4203408360481262, "rewards/meter/std": 0.4558182656764984, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9930034875869751, "rewards/repeat_soft/std": 0.014712625183165073, "rewards/judge_quality/mean": 0.5099999904632568, "rewards/judge_quality/std": 0.1302744597196579, "rewards/total_composite/mean": 0.4715820252895355, "rewards/total_composite/std": 0.1289324164390564, "reward": 0.4715820252895355, "reward_std": 0.1289324015378952, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16170267760753632, "sampling/sampling_logp_difference/max": 2.8092567920684814, "sampling/importance_sampling_ratio/min": 0.060249753296375275, "sampling/importance_sampling_ratio/mean": 1.0188674926757812, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0982524752616882, "clip_ratio/low_mean": 0.0752681796438992, "clip_ratio/low_min": 0.0752681796438992, "clip_ratio/high_mean": 0.06160464696586132, "clip_ratio/high_max": 0.06160464696586132, "clip_ratio/region_mean": 0.13687282660976052, "reward_total_mean": 0.4715820252895355, "reward_meter_mean": 0.4203408360481262, "reward_meter_std": 0.4558182656764984, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9930034875869751, "reward_repeat_soft_std": 0.014712625183165073, "reward_judge_quality_mean": 0.5099999904632568, "reward_judge_quality_std": 0.1302744597196579, "reward_total_composite_mean": 0.4715820252895355, "reward_total_composite_std": 0.1289324164390564} {"timestamp_utc": "2026-04-13T08:19:21Z", "mode": "train", "global_step": 388, "epoch": 0.038975389251632346, "loss": -0.1057, "grad_norm": 2.612093448638916, "learning_rate": 8.827272727272727e-06, "num_tokens": 703180.0, "completions/mean_length": 95.0, "completions/min_length": 33.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 35.42857360839844, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.9530330896377563, "rewards/meter/std": 0.08797551691532135, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9803513288497925, "rewards/repeat_soft/std": 0.03263489902019501, "rewards/judge_quality/mean": 0.6299999952316284, "rewards/judge_quality/std": 0.2958281338214874, "rewards/total_composite/mean": 0.7036803364753723, "rewards/total_composite/std": 0.3078506886959076, "reward": 0.7036803364753723, "reward_std": 0.3078506886959076, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21610069274902344, "sampling/sampling_logp_difference/max": 1.2582550048828125, "sampling/importance_sampling_ratio/min": 0.2841494381427765, "sampling/importance_sampling_ratio/mean": 1.0802019834518433, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3301824927330017, "clip_ratio/low_mean": 0.04922780115157366, "clip_ratio/low_min": 0.04922780115157366, "clip_ratio/high_mean": 0.10317971557378769, "clip_ratio/high_max": 0.10317971557378769, "clip_ratio/region_mean": 0.15240751672536135, "reward_total_mean": 0.7036803364753723, "reward_meter_mean": 0.9530330896377563, "reward_meter_std": 0.08797551691532135, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9803513288497925, "reward_repeat_soft_std": 0.03263489902019501, "reward_judge_quality_mean": 0.6299999952316284, "reward_judge_quality_std": 0.2958281338214874, "reward_total_composite_mean": 0.7036803364753723, "reward_total_composite_std": 0.3078506886959076} {"timestamp_utc": "2026-04-13T08:19:33Z", "mode": "train", "global_step": 389, "epoch": 0.039075841285786034, "loss": -0.0934, "grad_norm": 5.493145942687988, "learning_rate": 8.824242424242426e-06, "num_tokens": 705116.0, "completions/mean_length": 129.0, "completions/min_length": 68.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 74.28572082519531, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.8792860507965088, "rewards/meter/std": 0.23752202093601227, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9941399097442627, "rewards/repeat_soft/std": 0.0030691882129758596, "rewards/judge_quality/mean": 0.3687500059604645, "rewards/judge_quality/std": 0.1940867006778717, "rewards/total_composite/mean": 0.43528708815574646, "rewards/total_composite/std": 0.2920048236846924, "reward": 0.43528708815574646, "reward_std": 0.2920048236846924, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23220102488994598, "sampling/sampling_logp_difference/max": 1.0492897033691406, "sampling/importance_sampling_ratio/min": 0.35018637776374817, "sampling/importance_sampling_ratio/mean": 1.0767548084259033, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.9076579213142395, "clip_ratio/low_mean": 0.04516545124351978, "clip_ratio/low_min": 0.04516545124351978, "clip_ratio/high_mean": 0.09360717888921499, "clip_ratio/high_max": 0.09360717888921499, "clip_ratio/region_mean": 0.13877263013273478, "reward_total_mean": 0.43528708815574646, "reward_meter_mean": 0.8792860507965088, "reward_meter_std": 0.23752202093601227, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9941399097442627, "reward_repeat_soft_std": 0.0030691882129758596, "reward_judge_quality_mean": 0.3687500059604645, "reward_judge_quality_std": 0.1940867006778717, "reward_total_composite_mean": 0.43528708815574646, "reward_total_composite_std": 0.2920048236846924} {"timestamp_utc": "2026-04-13T08:19:40Z", "mode": "train", "global_step": 390, "epoch": 0.03917629331993973, "loss": 0.0821, "grad_norm": 15.537460327148438, "learning_rate": 8.821212121212121e-06, "num_tokens": 707297.0, "completions/mean_length": 82.625, "completions/min_length": 73.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.625, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.6493761539459229, "rewards/meter/std": 0.27126702666282654, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9950517416000366, "rewards/repeat_soft/std": 0.0026858015917241573, "rewards/judge_quality/mean": 0.6574999690055847, "rewards/judge_quality/std": 0.1505940854549408, "rewards/total_composite/mean": 0.6419288516044617, "rewards/total_composite/std": 0.1497366577386856, "reward": 0.6419288516044617, "reward_std": 0.1497366577386856, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23744165897369385, "sampling/sampling_logp_difference/max": 3.6390271186828613, "sampling/importance_sampling_ratio/min": 0.026277896016836166, "sampling/importance_sampling_ratio/mean": 0.9899482727050781, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4409097135066986, "clip_ratio/low_mean": 0.1076983530074358, "clip_ratio/low_min": 0.1076983530074358, "clip_ratio/high_mean": 0.10639841854572296, "clip_ratio/high_max": 0.10639841854572296, "clip_ratio/region_mean": 0.21409677155315876, "reward_total_mean": 0.6419288516044617, "reward_meter_mean": 0.6493761539459229, "reward_meter_std": 0.27126702666282654, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9950517416000366, "reward_repeat_soft_std": 0.0026858015917241573, "reward_judge_quality_mean": 0.6574999690055847, "reward_judge_quality_std": 0.1505940854549408, "reward_total_composite_mean": 0.6419288516044617, "reward_total_composite_std": 0.1497366577386856} {"timestamp_utc": "2026-04-13T08:19:46Z", "mode": "train", "global_step": 391, "epoch": 0.03927674535409342, "loss": 0.0345, "grad_norm": 19.484699249267578, "learning_rate": 8.818181818181819e-06, "num_tokens": 708686.0, "completions/mean_length": 33.625, "completions/min_length": 28.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.625, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.6881487965583801, "rewards/meter/std": 0.4074837863445282, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9906495213508606, "rewards/repeat_soft/std": 0.014289024285972118, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.5396751165390015, "rewards/total_composite/std": 0.1060907319188118, "reward": 0.5396751165390015, "reward_std": 0.1060907244682312, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19514882564544678, "sampling/sampling_logp_difference/max": 1.788926601409912, "sampling/importance_sampling_ratio/min": 0.16713947057724, "sampling/importance_sampling_ratio/mean": 1.0347812175750732, "sampling/importance_sampling_ratio/max": 1.9738614559173584, "entropy": 1.9678066968917847, "clip_ratio/low_mean": 0.055244164541363716, "clip_ratio/low_min": 0.055244164541363716, "clip_ratio/high_mean": 0.14656863175332546, "clip_ratio/high_max": 0.14656863175332546, "clip_ratio/region_mean": 0.20181279629468918, "reward_total_mean": 0.5396751165390015, "reward_meter_mean": 0.6881487965583801, "reward_meter_std": 0.4074837863445282, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9906495213508606, "reward_repeat_soft_std": 0.014289024285972118, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.5396751165390015, "reward_total_composite_std": 0.1060907319188118} {"timestamp_utc": "2026-04-13T08:19:53Z", "mode": "train", "global_step": 392, "epoch": 0.03937719738824711, "loss": -0.0108, "grad_norm": 13.432182312011719, "learning_rate": 8.815151515151516e-06, "num_tokens": 710620.0, "completions/mean_length": 65.75, "completions/min_length": 46.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.75, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.17572955787181854, "rewards/meter/std": 0.13979798555374146, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9765012264251709, "rewards/repeat_soft/std": 0.026776380836963654, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.39454472064971924, "rewards/total_composite/std": 0.03480697423219681, "reward": 0.39454472064971924, "reward_std": 0.03480697423219681, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21580170094966888, "sampling/sampling_logp_difference/max": 1.3005542755126953, "sampling/importance_sampling_ratio/min": 0.2723807692527771, "sampling/importance_sampling_ratio/mean": 1.0618659257888794, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.9697498977184296, "clip_ratio/low_mean": 0.05944700539112091, "clip_ratio/low_min": 0.05944700539112091, "clip_ratio/high_mean": 0.1420013289898634, "clip_ratio/high_max": 0.1420013289898634, "clip_ratio/region_mean": 0.2014483343809843, "reward_total_mean": 0.39454472064971924, "reward_meter_mean": 0.17572955787181854, "reward_meter_std": 0.13979798555374146, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9765012264251709, "reward_repeat_soft_std": 0.026776380836963654, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.39454472064971924, "reward_total_composite_std": 0.03480697423219681} {"timestamp_utc": "2026-04-13T08:20:00Z", "mode": "train", "global_step": 393, "epoch": 0.039477649422400805, "loss": 0.1136, "grad_norm": 13.407106399536133, "learning_rate": 8.812121212121213e-06, "num_tokens": 712500.0, "completions/mean_length": 68.0, "completions/min_length": 56.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.0, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.56876540184021, "rewards/meter/std": 0.38598161935806274, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9751752614974976, "rewards/repeat_soft/std": 0.026960745453834534, "rewards/judge_quality/mean": 0.4737499952316284, "rewards/judge_quality/std": 0.16291432082653046, "rewards/total_composite/mean": 0.5403320789337158, "rewards/total_composite/std": 0.17929324507713318, "reward": 0.5403320789337158, "reward_std": 0.17929324507713318, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22484685480594635, "sampling/sampling_logp_difference/max": 1.744100570678711, "sampling/importance_sampling_ratio/min": 0.17480213940143585, "sampling/importance_sampling_ratio/mean": 1.0354881286621094, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3533735424280167, "clip_ratio/low_mean": 0.1128325741738081, "clip_ratio/low_min": 0.1128325741738081, "clip_ratio/high_mean": 0.05654893722385168, "clip_ratio/high_max": 0.05654893722385168, "clip_ratio/region_mean": 0.16938151139765978, "reward_total_mean": 0.5403320789337158, "reward_meter_mean": 0.56876540184021, "reward_meter_std": 0.38598161935806274, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9751752614974976, "reward_repeat_soft_std": 0.026960745453834534, "reward_judge_quality_mean": 0.4737499952316284, "reward_judge_quality_std": 0.16291432082653046, "reward_total_composite_mean": 0.5403320789337158, "reward_total_composite_std": 0.17929324507713318} {"timestamp_utc": "2026-04-13T08:20:07Z", "mode": "train", "global_step": 394, "epoch": 0.03957810145655449, "loss": 0.0969, "grad_norm": 14.463814735412598, "learning_rate": 8.809090909090909e-06, "num_tokens": 714150.0, "completions/mean_length": 39.25, "completions/min_length": 31.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.25, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.49386802315711975, "rewards/meter/std": 0.3387821912765503, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.992102861404419, "rewards/repeat_soft/std": 0.008454841561615467, "rewards/judge_quality/mean": 0.5475000143051147, "rewards/judge_quality/std": 0.15191635489463806, "rewards/total_composite/mean": 0.5453914403915405, "rewards/total_composite/std": 0.1739407628774643, "reward": 0.5453914403915405, "reward_std": 0.1739407628774643, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18999524414539337, "sampling/sampling_logp_difference/max": 2.017080307006836, "sampling/importance_sampling_ratio/min": 0.1330433487892151, "sampling/importance_sampling_ratio/mean": 1.034308910369873, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5738208591938019, "clip_ratio/low_mean": 0.12540349457412958, "clip_ratio/low_min": 0.12540349457412958, "clip_ratio/high_mean": 0.04475806374102831, "clip_ratio/high_max": 0.04475806374102831, "clip_ratio/region_mean": 0.1701615583151579, "reward_total_mean": 0.5453914403915405, "reward_meter_mean": 0.49386802315711975, "reward_meter_std": 0.3387821912765503, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.992102861404419, "reward_repeat_soft_std": 0.008454841561615467, "reward_judge_quality_mean": 0.5475000143051147, "reward_judge_quality_std": 0.15191635489463806, "reward_total_composite_mean": 0.5453914403915405, "reward_total_composite_std": 0.1739407628774643} {"timestamp_utc": "2026-04-13T08:20:19Z", "mode": "train", "global_step": 395, "epoch": 0.03967855349070819, "loss": -0.0986, "grad_norm": 5.845329284667969, "learning_rate": 8.806060606060608e-06, "num_tokens": 716294.0, "completions/mean_length": 158.0, "completions/min_length": 86.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 107.42857360839844, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 127.0, "rewards/meter/mean": 0.8389735221862793, "rewards/meter/std": 0.2774248421192169, "rewards/count_adherence/mean": 0.8500000238418579, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.976121723651886, "rewards/repeat_soft/std": 0.017020486295223236, "rewards/judge_quality/mean": 0.3187499940395355, "rewards/judge_quality/std": 0.14961259067058563, "rewards/total_composite/mean": 0.45580732822418213, "rewards/total_composite/std": 0.20585764944553375, "reward": 0.45580732822418213, "reward_std": 0.20585763454437256, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2123926728963852, "sampling/sampling_logp_difference/max": 1.4099535942077637, "sampling/importance_sampling_ratio/min": 0.24415461719036102, "sampling/importance_sampling_ratio/mean": 1.0625028610229492, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.605967551469803, "clip_ratio/low_mean": 0.044079553335905075, "clip_ratio/low_min": 0.044079553335905075, "clip_ratio/high_mean": 0.09661378152668476, "clip_ratio/high_max": 0.09661378152668476, "clip_ratio/region_mean": 0.14069333486258984, "reward_total_mean": 0.45580732822418213, "reward_meter_mean": 0.8389735221862793, "reward_meter_std": 0.2774248421192169, "reward_count_adherence_mean": 0.8500000238418579, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.976121723651886, "reward_repeat_soft_std": 0.017020486295223236, "reward_judge_quality_mean": 0.3187499940395355, "reward_judge_quality_std": 0.14961259067058563, "reward_total_composite_mean": 0.45580732822418213, "reward_total_composite_std": 0.20585764944553375} {"timestamp_utc": "2026-04-13T08:20:28Z", "mode": "train", "global_step": 396, "epoch": 0.039779005524861875, "loss": 0.3118, "grad_norm": 10.84103012084961, "learning_rate": 8.803030303030303e-06, "num_tokens": 718666.0, "completions/mean_length": 107.5, "completions/min_length": 70.0, "completions/max_length": 221.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.5, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 221.0, "rewards/meter/mean": 0.6435260772705078, "rewards/meter/std": 0.40282678604125977, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.21380899846553802, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9662259817123413, "rewards/repeat_soft/std": 0.02086596004664898, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.2121320515871048, "rewards/total_composite/mean": 0.447258323431015, "rewards/total_composite/std": 0.31164485216140747, "reward": 0.447258323431015, "reward_std": 0.31164485216140747, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21521738171577454, "sampling/sampling_logp_difference/max": 1.9176673889160156, "sampling/importance_sampling_ratio/min": 0.14694933593273163, "sampling/importance_sampling_ratio/mean": 1.041980504989624, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.2222574949264526, "clip_ratio/low_mean": 0.05741437803953886, "clip_ratio/low_min": 0.05741437803953886, "clip_ratio/high_mean": 0.11796465516090393, "clip_ratio/high_max": 0.11796465516090393, "clip_ratio/region_mean": 0.1753790332004428, "reward_total_mean": 0.447258323431015, "reward_meter_mean": 0.6435260772705078, "reward_meter_std": 0.40282678604125977, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.21380899846553802, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9662259817123413, "reward_repeat_soft_std": 0.02086596004664898, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.2121320515871048, "reward_total_composite_mean": 0.447258323431015, "reward_total_composite_std": 0.31164485216140747} {"timestamp_utc": "2026-04-13T08:20:35Z", "mode": "train", "global_step": 397, "epoch": 0.03987945755901557, "loss": -0.029, "grad_norm": 13.39407730102539, "learning_rate": 8.8e-06, "num_tokens": 720361.0, "completions/mean_length": 40.875, "completions/min_length": 36.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.875, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.7563626766204834, "rewards/meter/std": 0.34741678833961487, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9957313537597656, "rewards/repeat_soft/std": 0.004118890967220068, "rewards/judge_quality/mean": 0.6474999785423279, "rewards/judge_quality/std": 0.2365073412656784, "rewards/total_composite/mean": 0.6738553047180176, "rewards/total_composite/std": 0.18831169605255127, "reward": 0.6738553047180176, "reward_std": 0.18831169605255127, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1969454139471054, "sampling/sampling_logp_difference/max": 1.505446434020996, "sampling/importance_sampling_ratio/min": 0.22191819548606873, "sampling/importance_sampling_ratio/mean": 1.0099977254867554, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6303172707557678, "clip_ratio/low_mean": 0.09531616419553757, "clip_ratio/low_min": 0.09531616419553757, "clip_ratio/high_mean": 0.06741962395608425, "clip_ratio/high_max": 0.06741962395608425, "clip_ratio/region_mean": 0.16273578815162182, "reward_total_mean": 0.6738553047180176, "reward_meter_mean": 0.7563626766204834, "reward_meter_std": 0.34741678833961487, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9957313537597656, "reward_repeat_soft_std": 0.004118890967220068, "reward_judge_quality_mean": 0.6474999785423279, "reward_judge_quality_std": 0.2365073412656784, "reward_total_composite_mean": 0.6738553047180176, "reward_total_composite_std": 0.18831169605255127} {"timestamp_utc": "2026-04-13T08:20:42Z", "mode": "train", "global_step": 398, "epoch": 0.039979909593169265, "loss": 0.1119, "grad_norm": 18.618391036987305, "learning_rate": 8.796969696969698e-06, "num_tokens": 722027.0, "completions/mean_length": 39.25, "completions/min_length": 30.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.25, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.692610502243042, "rewards/meter/std": 0.39874544739723206, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9921975135803223, "rewards/repeat_soft/std": 0.006021677982062101, "rewards/judge_quality/mean": 0.6449999809265137, "rewards/judge_quality/std": 0.24928471446037292, "rewards/total_composite/mean": 0.6342549324035645, "rewards/total_composite/std": 0.20082521438598633, "reward": 0.6342549324035645, "reward_std": 0.20082521438598633, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2037847340106964, "sampling/sampling_logp_difference/max": 1.1941423416137695, "sampling/importance_sampling_ratio/min": 0.3029636740684509, "sampling/importance_sampling_ratio/mean": 1.0290169715881348, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.982935443520546, "clip_ratio/low_mean": 0.12808520812541246, "clip_ratio/low_min": 0.12808520812541246, "clip_ratio/high_mean": 0.07797619327902794, "clip_ratio/high_max": 0.07797619327902794, "clip_ratio/region_mean": 0.2060614014044404, "reward_total_mean": 0.6342549324035645, "reward_meter_mean": 0.692610502243042, "reward_meter_std": 0.39874544739723206, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9921975135803223, "reward_repeat_soft_std": 0.006021677982062101, "reward_judge_quality_mean": 0.6449999809265137, "reward_judge_quality_std": 0.24928471446037292, "reward_total_composite_mean": 0.6342549324035645, "reward_total_composite_std": 0.20082521438598633} {"timestamp_utc": "2026-04-13T08:20:54Z", "mode": "train", "global_step": 399, "epoch": 0.04008036162732295, "loss": 0.1911, "grad_norm": 4.706899166107178, "learning_rate": 8.793939393939395e-06, "num_tokens": 724384.0, "completions/mean_length": 162.625, "completions/min_length": 72.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 112.71429443359375, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 309.0, "rewards/meter/mean": 0.6536551713943481, "rewards/meter/std": 0.36277490854263306, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9848779439926147, "rewards/repeat_soft/std": 0.028398148715496063, "rewards/judge_quality/mean": 0.32749998569488525, "rewards/judge_quality/std": 0.17127670347690582, "rewards/total_composite/mean": 0.4322788715362549, "rewards/total_composite/std": 0.2684842050075531, "reward": 0.4322788715362549, "reward_std": 0.2684842050075531, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19599947333335876, "sampling/sampling_logp_difference/max": 1.4805917739868164, "sampling/importance_sampling_ratio/min": 0.22750303149223328, "sampling/importance_sampling_ratio/mean": 1.030487298965454, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.985503315925598, "clip_ratio/low_mean": 0.007281553465873003, "clip_ratio/low_min": 0.007281553465873003, "clip_ratio/high_mean": 0.15667317807674408, "clip_ratio/high_max": 0.15667317807674408, "clip_ratio/region_mean": 0.16395473154261708, "reward_total_mean": 0.4322788715362549, "reward_meter_mean": 0.6536551713943481, "reward_meter_std": 0.36277490854263306, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9848779439926147, "reward_repeat_soft_std": 0.028398148715496063, "reward_judge_quality_mean": 0.32749998569488525, "reward_judge_quality_std": 0.17127670347690582, "reward_total_composite_mean": 0.4322788715362549, "reward_total_composite_std": 0.2684842050075531} {"timestamp_utc": "2026-04-13T08:21:01Z", "mode": "train", "global_step": 400, "epoch": 0.04018081366147665, "loss": 0.0069, "grad_norm": 17.07670021057129, "learning_rate": 8.790909090909092e-06, "num_tokens": 725874.0, "completions/mean_length": 34.25, "completions/min_length": 30.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.25, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.9133437871932983, "rewards/meter/std": 0.10935598611831665, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9907272458076477, "rewards/repeat_soft/std": 0.0059598698280751705, "rewards/judge_quality/mean": 0.7275000214576721, "rewards/judge_quality/std": 0.20119288563728333, "rewards/total_composite/mean": 0.7812926769256592, "rewards/total_composite/std": 0.13477379083633423, "reward": 0.7812926769256592, "reward_std": 0.13477377593517303, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15207621455192566, "sampling/sampling_logp_difference/max": 1.8708162307739258, "sampling/importance_sampling_ratio/min": 0.15399791300296783, "sampling/importance_sampling_ratio/mean": 1.025518774986267, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.907796710729599, "clip_ratio/low_mean": 0.044287471333518624, "clip_ratio/low_min": 0.044287471333518624, "clip_ratio/high_mean": 0.08598564472049475, "clip_ratio/high_max": 0.08598564472049475, "clip_ratio/region_mean": 0.13027311605401337, "reward_total_mean": 0.7812926769256592, "reward_meter_mean": 0.9133437871932983, "reward_meter_std": 0.10935598611831665, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9907272458076477, "reward_repeat_soft_std": 0.0059598698280751705, "reward_judge_quality_mean": 0.7275000214576721, "reward_judge_quality_std": 0.20119288563728333, "reward_total_composite_mean": 0.7812926769256592, "reward_total_composite_std": 0.13477379083633423} {"timestamp_utc": "2026-04-13T08:21:52Z", "mode": "eval", "global_step": 400, "epoch": 0.04018081366147665, "eval_loss": NaN, "eval_runtime": 50.4575, "eval_samples_per_second": 1.585, "eval_steps_per_second": 0.198, "eval_num_tokens": 725874.0, "eval_completions/mean_length": 71.0625, "eval_completions/min_length": 27.7, "eval_completions/max_length": 147.3, "eval_completions/clipped_ratio": 0.025, "eval_completions/mean_terminated_length": 59.775, "eval_completions/min_terminated_length": 27.7, "eval_completions/max_terminated_length": 104.7, "eval_rewards/meter/mean": 0.6084526628255844, "eval_rewards/meter/std": 0.3190243661403656, "eval_rewards/count_adherence/mean": 0.9627083301544189, "eval_rewards/count_adherence/std": 0.07116240337491035, "eval_rewards/hard_gate/mean": 0.9375, "eval_rewards/hard_gate/std": 0.1767766922712326, "eval_rewards/repeat_soft/mean": 0.976105797290802, "eval_rewards/repeat_soft/std": 0.028223066963255404, "eval_rewards/judge_quality/mean": 0.5106249988079071, "eval_rewards/judge_quality/std": 0.17371627036482096, "eval_rewards/total_composite/mean": 0.5150679022073745, "eval_rewards/total_composite/std": 0.17966578379273415, "eval_reward": 0.5150679022073745, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.13076039776206017, "eval_sampling/sampling_logp_difference/max": 1.2090730667114258, "eval_sampling/importance_sampling_ratio/min": 0.304891861975193, "eval_sampling/importance_sampling_ratio/mean": 1.0383795976638794, "eval_sampling/importance_sampling_ratio/max": 1.538509440422058, "eval_entropy": 1.8213319420814513, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5150679022073745, "eval_reward_meter_mean": 0.6084526628255844, "eval_reward_meter_std": 0.3190243661403656, "eval_reward_count_adherence_mean": 0.9627083301544189, "eval_reward_count_adherence_std": 0.07116240337491035, "eval_reward_hard_gate_mean": 0.9375, "eval_reward_hard_gate_std": 0.1767766922712326, "eval_reward_repeat_soft_mean": 0.976105797290802, "eval_reward_repeat_soft_std": 0.028223066963255404, "eval_reward_judge_quality_mean": 0.5106249988079071, "eval_reward_judge_quality_std": 0.17371627036482096, "eval_reward_total_composite_mean": 0.5150679022073745, "eval_reward_total_composite_std": 0.17966578379273415} {"timestamp_utc": "2026-04-13T08:22:03Z", "mode": "train", "global_step": 401, "epoch": 0.040281265695630335, "loss": 0.1463, "grad_norm": 13.34891128540039, "learning_rate": 8.787878787878788e-06, "num_tokens": 727854.0, "completions/mean_length": 67.5, "completions/min_length": 58.0, "completions/max_length": 97.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.5, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.3987351059913635, "rewards/meter/std": 0.24704766273498535, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9943833351135254, "rewards/repeat_soft/std": 0.003714403137564659, "rewards/judge_quality/mean": 0.6449999809265137, "rewards/judge_quality/std": 0.24928471446037292, "rewards/total_composite/mean": 0.5330898761749268, "rewards/total_composite/std": 0.15060415863990784, "reward": 0.5330898761749268, "reward_std": 0.15060415863990784, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19010724127292633, "sampling/sampling_logp_difference/max": 2.565248489379883, "sampling/importance_sampling_ratio/min": 0.07690006494522095, "sampling/importance_sampling_ratio/mean": 1.0112743377685547, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2907120138406754, "clip_ratio/low_mean": 0.10911228507757187, "clip_ratio/low_min": 0.10911228507757187, "clip_ratio/high_mean": 0.08028915338218212, "clip_ratio/high_max": 0.08028915338218212, "clip_ratio/region_mean": 0.189401438459754, "reward_total_mean": 0.5330898761749268, "reward_meter_mean": 0.3987351059913635, "reward_meter_std": 0.24704766273498535, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9943833351135254, "reward_repeat_soft_std": 0.003714403137564659, "reward_judge_quality_mean": 0.6449999809265137, "reward_judge_quality_std": 0.24928471446037292, "reward_total_composite_mean": 0.5330898761749268, "reward_total_composite_std": 0.15060415863990784} {"timestamp_utc": "2026-04-13T08:22:10Z", "mode": "train", "global_step": 402, "epoch": 0.04038171772978403, "loss": 0.0708, "grad_norm": 25.57284164428711, "learning_rate": 8.784848484848487e-06, "num_tokens": 729290.0, "completions/mean_length": 19.5, "completions/min_length": 14.0, "completions/max_length": 24.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 19.5, "completions/min_terminated_length": 14.0, "completions/max_terminated_length": 24.0, "rewards/meter/mean": 0.8696723580360413, "rewards/meter/std": 0.2797050476074219, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9484448432922363, "rewards/repeat_soft/std": 0.017298834398388863, "rewards/judge_quality/mean": 0.3087500035762787, "rewards/judge_quality/std": 0.1141975000500679, "rewards/total_composite/mean": 0.4683290421962738, "rewards/total_composite/std": 0.2080000638961792, "reward": 0.4683290421962738, "reward_std": 0.2080000638961792, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20655979216098785, "sampling/sampling_logp_difference/max": 1.509596347808838, "sampling/importance_sampling_ratio/min": 0.22099916636943817, "sampling/importance_sampling_ratio/mean": 1.0491156578063965, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2838480323553085, "clip_ratio/low_mean": 0.02878289483487606, "clip_ratio/low_min": 0.02878289483487606, "clip_ratio/high_mean": 0.14130889158695936, "clip_ratio/high_max": 0.14130889158695936, "clip_ratio/region_mean": 0.17009178642183542, "reward_total_mean": 0.4683290421962738, "reward_meter_mean": 0.8696723580360413, "reward_meter_std": 0.2797050476074219, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9484448432922363, "reward_repeat_soft_std": 0.017298834398388863, "reward_judge_quality_mean": 0.3087500035762787, "reward_judge_quality_std": 0.1141975000500679, "reward_total_composite_mean": 0.4683290421962738, "reward_total_composite_std": 0.2080000638961792} {"timestamp_utc": "2026-04-13T08:22:18Z", "mode": "train", "global_step": 403, "epoch": 0.04048216976393772, "loss": 0.4829, "grad_norm": 22.063720703125, "learning_rate": 8.781818181818182e-06, "num_tokens": 730781.0, "completions/mean_length": 26.375, "completions/min_length": 16.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.375, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.7853130102157593, "rewards/meter/std": 0.34553444385528564, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9394265413284302, "rewards/repeat_soft/std": 0.03469981253147125, "rewards/judge_quality/mean": 0.5737500190734863, "rewards/judge_quality/std": 0.3009004592895508, "rewards/total_composite/mean": 0.6728497743606567, "rewards/total_composite/std": 0.22580192983150482, "reward": 0.6728497743606567, "reward_std": 0.22580192983150482, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21233589947223663, "sampling/sampling_logp_difference/max": 1.7593927383422852, "sampling/importance_sampling_ratio/min": 0.17214937508106232, "sampling/importance_sampling_ratio/mean": 1.0368895530700684, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9786519557237625, "clip_ratio/low_mean": 0.15014297794550657, "clip_ratio/low_min": 0.15014297794550657, "clip_ratio/high_mean": 0.0657894741743803, "clip_ratio/high_max": 0.0657894741743803, "clip_ratio/region_mean": 0.21593245211988688, "reward_total_mean": 0.6728497743606567, "reward_meter_mean": 0.7853130102157593, "reward_meter_std": 0.34553444385528564, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9394265413284302, "reward_repeat_soft_std": 0.03469981253147125, "reward_judge_quality_mean": 0.5737500190734863, "reward_judge_quality_std": 0.3009004592895508, "reward_total_composite_mean": 0.6728497743606567, "reward_total_composite_std": 0.22580192983150482} {"timestamp_utc": "2026-04-13T08:22:25Z", "mode": "train", "global_step": 404, "epoch": 0.04058262179809141, "loss": 0.0177, "grad_norm": 9.488207817077637, "learning_rate": 8.77878787878788e-06, "num_tokens": 733016.0, "completions/mean_length": 81.375, "completions/min_length": 72.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 81.375, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.9766528606414795, "rewards/meter/std": 0.032082848250865936, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9504460096359253, "rewards/repeat_soft/std": 0.06312248110771179, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.6209337115287781, "rewards/total_composite/std": 0.0731566920876503, "reward": 0.6209337115287781, "reward_std": 0.07315670698881149, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18496696650981903, "sampling/sampling_logp_difference/max": 1.6487913131713867, "sampling/importance_sampling_ratio/min": 0.19228217005729675, "sampling/importance_sampling_ratio/mean": 1.0251444578170776, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8824598789215088, "clip_ratio/low_mean": 0.15241516008973122, "clip_ratio/low_min": 0.15241516008973122, "clip_ratio/high_mean": 0.017241379246115685, "clip_ratio/high_max": 0.017241379246115685, "clip_ratio/region_mean": 0.1696565393358469, "reward_total_mean": 0.6209337115287781, "reward_meter_mean": 0.9766528606414795, "reward_meter_std": 0.032082848250865936, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9504460096359253, "reward_repeat_soft_std": 0.06312248110771179, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.6209337115287781, "reward_total_composite_std": 0.0731566920876503} {"timestamp_utc": "2026-04-13T08:22:33Z", "mode": "train", "global_step": 405, "epoch": 0.040683073832245106, "loss": 0.059, "grad_norm": 13.412790298461914, "learning_rate": 8.775757575757577e-06, "num_tokens": 735115.0, "completions/mean_length": 78.375, "completions/min_length": 65.0, "completions/max_length": 95.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.375, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.589837908744812, "rewards/meter/std": 0.3526695668697357, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9877939224243164, "rewards/repeat_soft/std": 0.009391658008098602, "rewards/judge_quality/mean": 0.5324999690055847, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.5582143664360046, "rewards/total_composite/std": 0.14221405982971191, "reward": 0.5582143664360046, "reward_std": 0.14221405982971191, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18767502903938293, "sampling/sampling_logp_difference/max": 1.452690601348877, "sampling/importance_sampling_ratio/min": 0.2339400053024292, "sampling/importance_sampling_ratio/mean": 1.0276070833206177, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6444737315177917, "clip_ratio/low_mean": 0.07759459409862757, "clip_ratio/low_min": 0.07759459409862757, "clip_ratio/high_mean": 0.07964119128882885, "clip_ratio/high_max": 0.07964119128882885, "clip_ratio/region_mean": 0.15723578538745642, "reward_total_mean": 0.5582143664360046, "reward_meter_mean": 0.589837908744812, "reward_meter_std": 0.3526695668697357, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9877939224243164, "reward_repeat_soft_std": 0.009391658008098602, "reward_judge_quality_mean": 0.5324999690055847, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.5582143664360046, "reward_total_composite_std": 0.14221405982971191} {"timestamp_utc": "2026-04-13T08:22:40Z", "mode": "train", "global_step": 406, "epoch": 0.040783525866398794, "loss": 0.1202, "grad_norm": 18.717052459716797, "learning_rate": 8.772727272727274e-06, "num_tokens": 736681.0, "completions/mean_length": 40.75, "completions/min_length": 33.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.75, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.9577003717422485, "rewards/meter/std": 0.058974817395210266, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9921804666519165, "rewards/repeat_soft/std": 0.0125203812494874, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.6150851249694824, "rewards/total_composite/std": 0.02064206451177597, "reward": 0.6150851249694824, "reward_std": 0.020642060786485672, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2295956015586853, "sampling/sampling_logp_difference/max": 1.4451169967651367, "sampling/importance_sampling_ratio/min": 0.2357185035943985, "sampling/importance_sampling_ratio/mean": 1.0493429899215698, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.5007997155189514, "clip_ratio/low_mean": 0.06051027122884989, "clip_ratio/low_min": 0.06051027122884989, "clip_ratio/high_mean": 0.13226395472884178, "clip_ratio/high_max": 0.13226395472884178, "clip_ratio/region_mean": 0.19277422595769167, "reward_total_mean": 0.6150851249694824, "reward_meter_mean": 0.9577003717422485, "reward_meter_std": 0.058974817395210266, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9921804666519165, "reward_repeat_soft_std": 0.0125203812494874, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.6150851249694824, "reward_total_composite_std": 0.02064206451177597} {"timestamp_utc": "2026-04-13T08:22:50Z", "mode": "train", "global_step": 407, "epoch": 0.04088397790055249, "loss": 0.0501, "grad_norm": 11.843538284301758, "learning_rate": 8.76969696969697e-06, "num_tokens": 738337.0, "completions/mean_length": 37.0, "completions/min_length": 31.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.0, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.17129772901535034, "rewards/meter/std": 0.25353488326072693, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9954594373703003, "rewards/repeat_soft/std": 0.0074075618758797646, "rewards/judge_quality/mean": 0.6112500429153442, "rewards/judge_quality/std": 0.25587037205696106, "rewards/total_composite/mean": 0.4334220290184021, "rewards/total_composite/std": 0.1568155437707901, "reward": 0.4334220290184021, "reward_std": 0.1568155437707901, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18124893307685852, "sampling/sampling_logp_difference/max": 1.4005489349365234, "sampling/importance_sampling_ratio/min": 0.2464616298675537, "sampling/importance_sampling_ratio/mean": 1.0053983926773071, "sampling/importance_sampling_ratio/max": 1.9805840253829956, "entropy": 1.377250149846077, "clip_ratio/low_mean": 0.16454838402569294, "clip_ratio/low_min": 0.16454838402569294, "clip_ratio/high_mean": 0.025735294446349144, "clip_ratio/high_max": 0.025735294446349144, "clip_ratio/region_mean": 0.19028367847204208, "reward_total_mean": 0.4334220290184021, "reward_meter_mean": 0.17129772901535034, "reward_meter_std": 0.25353488326072693, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9954594373703003, "reward_repeat_soft_std": 0.0074075618758797646, "reward_judge_quality_mean": 0.6112500429153442, "reward_judge_quality_std": 0.25587037205696106, "reward_total_composite_mean": 0.4334220290184021, "reward_total_composite_std": 0.1568155437707901} {"timestamp_utc": "2026-04-13T08:22:57Z", "mode": "train", "global_step": 408, "epoch": 0.040984429934706176, "loss": 0.0024, "grad_norm": 19.389144897460938, "learning_rate": 8.766666666666669e-06, "num_tokens": 740026.0, "completions/mean_length": 36.125, "completions/min_length": 30.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.125, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.8589898347854614, "rewards/meter/std": 0.22964926064014435, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9893051981925964, "rewards/repeat_soft/std": 0.01291816309094429, "rewards/judge_quality/mean": 0.4025000035762787, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.5730451345443726, "rewards/total_composite/std": 0.07296156883239746, "reward": 0.5730451345443726, "reward_std": 0.07296158373355865, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24116696417331696, "sampling/sampling_logp_difference/max": 1.9932687282562256, "sampling/importance_sampling_ratio/min": 0.1362493336200714, "sampling/importance_sampling_ratio/mean": 1.0360838174819946, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6887087672948837, "clip_ratio/low_mean": 0.07655342854559422, "clip_ratio/low_min": 0.07655342854559422, "clip_ratio/high_mean": 0.129474725574255, "clip_ratio/high_max": 0.129474725574255, "clip_ratio/region_mean": 0.2060281541198492, "reward_total_mean": 0.5730451345443726, "reward_meter_mean": 0.8589898347854614, "reward_meter_std": 0.22964926064014435, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9893051981925964, "reward_repeat_soft_std": 0.01291816309094429, "reward_judge_quality_mean": 0.4025000035762787, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.5730451345443726, "reward_total_composite_std": 0.07296156883239746} {"timestamp_utc": "2026-04-13T08:23:05Z", "mode": "train", "global_step": 409, "epoch": 0.04108488196885987, "loss": 0.0562, "grad_norm": 11.353716850280762, "learning_rate": 8.763636363636364e-06, "num_tokens": 742081.0, "completions/mean_length": 69.875, "completions/min_length": 62.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.875, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.8985353112220764, "rewards/meter/std": 0.22408749163150787, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9455304145812988, "rewards/repeat_soft/std": 0.06894509494304657, "rewards/judge_quality/mean": 0.5324999690055847, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.6432985067367554, "rewards/total_composite/std": 0.10527710616588593, "reward": 0.6432985067367554, "reward_std": 0.10527710616588593, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16822470724582672, "sampling/sampling_logp_difference/max": 2.2970826625823975, "sampling/importance_sampling_ratio/min": 0.10055176168680191, "sampling/importance_sampling_ratio/mean": 1.0406503677368164, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5529555454850197, "clip_ratio/low_mean": 0.132304847240448, "clip_ratio/low_min": 0.132304847240448, "clip_ratio/high_mean": 0.0498511902987957, "clip_ratio/high_max": 0.0498511902987957, "clip_ratio/region_mean": 0.1821560375392437, "reward_total_mean": 0.6432985067367554, "reward_meter_mean": 0.8985353112220764, "reward_meter_std": 0.22408749163150787, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9455304145812988, "reward_repeat_soft_std": 0.06894509494304657, "reward_judge_quality_mean": 0.5324999690055847, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.6432985067367554, "reward_total_composite_std": 0.10527710616588593} {"timestamp_utc": "2026-04-13T08:23:12Z", "mode": "train", "global_step": 410, "epoch": 0.04118533400301356, "loss": -0.0049, "grad_norm": 15.925487518310547, "learning_rate": 8.760606060606061e-06, "num_tokens": 743731.0, "completions/mean_length": 41.25, "completions/min_length": 35.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.25, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.6764265894889832, "rewards/meter/std": 0.39743971824645996, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9897750616073608, "rewards/repeat_soft/std": 0.017793547362089157, "rewards/judge_quality/mean": 0.6100000143051147, "rewards/judge_quality/std": 0.2093527615070343, "rewards/total_composite/mean": 0.6413960456848145, "rewards/total_composite/std": 0.21809059381484985, "reward": 0.6413960456848145, "reward_std": 0.21809059381484985, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17137528955936432, "sampling/sampling_logp_difference/max": 1.964127540588379, "sampling/importance_sampling_ratio/min": 0.14027822017669678, "sampling/importance_sampling_ratio/mean": 1.0164107084274292, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1172003895044327, "clip_ratio/low_mean": 0.0888611413538456, "clip_ratio/low_min": 0.0888611413538456, "clip_ratio/high_mean": 0.06415513809770346, "clip_ratio/high_max": 0.06415513809770346, "clip_ratio/region_mean": 0.15301627945154905, "reward_total_mean": 0.6413960456848145, "reward_meter_mean": 0.6764265894889832, "reward_meter_std": 0.39743971824645996, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9897750616073608, "reward_repeat_soft_std": 0.017793547362089157, "reward_judge_quality_mean": 0.6100000143051147, "reward_judge_quality_std": 0.2093527615070343, "reward_total_composite_mean": 0.6413960456848145, "reward_total_composite_std": 0.21809059381484985} {"timestamp_utc": "2026-04-13T08:23:20Z", "mode": "train", "global_step": 411, "epoch": 0.04128578603716725, "loss": 0.0128, "grad_norm": 18.98621368408203, "learning_rate": 8.757575757575759e-06, "num_tokens": 745330.0, "completions/mean_length": 39.875, "completions/min_length": 31.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.875, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.855193018913269, "rewards/meter/std": 0.27002963423728943, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9820802211761475, "rewards/repeat_soft/std": 0.020501619204878807, "rewards/judge_quality/mean": 0.6112499833106995, "rewards/judge_quality/std": 0.15037456154823303, "rewards/total_composite/mean": 0.5779845714569092, "rewards/total_composite/std": 0.261025607585907, "reward": 0.5779845714569092, "reward_std": 0.261025607585907, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1591351330280304, "sampling/sampling_logp_difference/max": 1.2452759742736816, "sampling/importance_sampling_ratio/min": 0.2878614664077759, "sampling/importance_sampling_ratio/mean": 1.024895191192627, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3111989423632622, "clip_ratio/low_mean": 0.04879032168537378, "clip_ratio/low_min": 0.04879032168537378, "clip_ratio/high_mean": 0.12211860902607441, "clip_ratio/high_max": 0.12211860902607441, "clip_ratio/region_mean": 0.1709089307114482, "reward_total_mean": 0.5779845714569092, "reward_meter_mean": 0.855193018913269, "reward_meter_std": 0.27002963423728943, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9820802211761475, "reward_repeat_soft_std": 0.020501619204878807, "reward_judge_quality_mean": 0.6112499833106995, "reward_judge_quality_std": 0.15037456154823303, "reward_total_composite_mean": 0.5779845714569092, "reward_total_composite_std": 0.261025607585907} {"timestamp_utc": "2026-04-13T08:23:29Z", "mode": "train", "global_step": 412, "epoch": 0.04138623807132094, "loss": -0.0314, "grad_norm": 13.070358276367188, "learning_rate": 8.754545454545456e-06, "num_tokens": 746713.0, "completions/mean_length": 32.875, "completions/min_length": 26.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.875, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.8472753763198853, "rewards/meter/std": 0.17025259137153625, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9989843964576721, "rewards/repeat_soft/std": 0.0015448419144377112, "rewards/judge_quality/mean": 0.6274999976158142, "rewards/judge_quality/std": 0.21952873468399048, "rewards/total_composite/mean": 0.6450576782226562, "rewards/total_composite/std": 0.3009953796863556, "reward": 0.6450576782226562, "reward_std": 0.3009953796863556, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15432707965373993, "sampling/sampling_logp_difference/max": 1.2746858596801758, "sampling/importance_sampling_ratio/min": 0.2795187532901764, "sampling/importance_sampling_ratio/mean": 1.008492112159729, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0406625121831894, "clip_ratio/low_mean": 0.05918387370184064, "clip_ratio/low_min": 0.05918387370184064, "clip_ratio/high_mean": 0.07324465177953243, "clip_ratio/high_max": 0.07324465177953243, "clip_ratio/region_mean": 0.13242852548137307, "reward_total_mean": 0.6450576782226562, "reward_meter_mean": 0.8472753763198853, "reward_meter_std": 0.17025259137153625, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9989843964576721, "reward_repeat_soft_std": 0.0015448419144377112, "reward_judge_quality_mean": 0.6274999976158142, "reward_judge_quality_std": 0.21952873468399048, "reward_total_composite_mean": 0.6450576782226562, "reward_total_composite_std": 0.3009953796863556} {"timestamp_utc": "2026-04-13T08:23:39Z", "mode": "train", "global_step": 413, "epoch": 0.041486690105474636, "loss": 0.0681, "grad_norm": 14.127117156982422, "learning_rate": 8.751515151515151e-06, "num_tokens": 748539.0, "completions/mean_length": 52.25, "completions/min_length": 46.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.25, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.8255071640014648, "rewards/meter/std": 0.2958468496799469, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9425725936889648, "rewards/repeat_soft/std": 0.03496149182319641, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.6523831486701965, "rewards/total_composite/std": 0.18113821744918823, "reward": 0.6523831486701965, "reward_std": 0.18113821744918823, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19584651291370392, "sampling/sampling_logp_difference/max": 1.7609224319458008, "sampling/importance_sampling_ratio/min": 0.17188623547554016, "sampling/importance_sampling_ratio/mean": 1.0101100206375122, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2705471515655518, "clip_ratio/low_mean": 0.0991121381521225, "clip_ratio/low_min": 0.0991121381521225, "clip_ratio/high_mean": 0.058673469349741936, "clip_ratio/high_max": 0.058673469349741936, "clip_ratio/region_mean": 0.15778560750186443, "reward_total_mean": 0.6523831486701965, "reward_meter_mean": 0.8255071640014648, "reward_meter_std": 0.2958468496799469, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9425725936889648, "reward_repeat_soft_std": 0.03496149182319641, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.6523831486701965, "reward_total_composite_std": 0.18113821744918823} {"timestamp_utc": "2026-04-13T08:23:46Z", "mode": "train", "global_step": 414, "epoch": 0.04158714213962833, "loss": 0.0386, "grad_norm": 19.73372459411621, "learning_rate": 8.748484848484849e-06, "num_tokens": 750014.0, "completions/mean_length": 29.375, "completions/min_length": 16.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.375, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.5379759073257446, "rewards/meter/std": 0.34564530849456787, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9569318294525146, "rewards/repeat_soft/std": 0.015749182552099228, "rewards/judge_quality/mean": 0.36375001072883606, "rewards/judge_quality/std": 0.13265825808048248, "rewards/total_composite/mean": 0.4616193175315857, "rewards/total_composite/std": 0.09335094690322876, "reward": 0.4616193175315857, "reward_std": 0.09335093945264816, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1824444681406021, "sampling/sampling_logp_difference/max": 1.346390724182129, "sampling/importance_sampling_ratio/min": 0.2601776123046875, "sampling/importance_sampling_ratio/mean": 1.032379150390625, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6422661319375038, "clip_ratio/low_mean": 0.08975493349134922, "clip_ratio/low_min": 0.08975493349134922, "clip_ratio/high_mean": 0.054734849371016026, "clip_ratio/high_max": 0.054734849371016026, "clip_ratio/region_mean": 0.14448978286236525, "reward_total_mean": 0.4616193175315857, "reward_meter_mean": 0.5379759073257446, "reward_meter_std": 0.34564530849456787, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9569318294525146, "reward_repeat_soft_std": 0.015749182552099228, "reward_judge_quality_mean": 0.36375001072883606, "reward_judge_quality_std": 0.13265825808048248, "reward_total_composite_mean": 0.4616193175315857, "reward_total_composite_std": 0.09335094690322876} {"timestamp_utc": "2026-04-13T08:23:52Z", "mode": "train", "global_step": 415, "epoch": 0.04168759417378202, "loss": 0.0945, "grad_norm": 22.352832794189453, "learning_rate": 8.745454545454546e-06, "num_tokens": 751704.0, "completions/mean_length": 47.25, "completions/min_length": 44.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.25, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.7549527883529663, "rewards/meter/std": 0.26335862278938293, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9893888235092163, "rewards/repeat_soft/std": 0.007539966143667698, "rewards/judge_quality/mean": 0.6737500429153442, "rewards/judge_quality/std": 0.263435423374176, "rewards/total_composite/mean": 0.6675245761871338, "rewards/total_composite/std": 0.15883620083332062, "reward": 0.6675245761871338, "reward_std": 0.15883618593215942, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24240705370903015, "sampling/sampling_logp_difference/max": 1.9810259342193604, "sampling/importance_sampling_ratio/min": 0.13792765140533447, "sampling/importance_sampling_ratio/mean": 1.0085594654083252, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3290906995534897, "clip_ratio/low_mean": 0.15271147526800632, "clip_ratio/low_min": 0.15271147526800632, "clip_ratio/high_mean": 0.06439394131302834, "clip_ratio/high_max": 0.06439394131302834, "clip_ratio/region_mean": 0.21710541658103466, "reward_total_mean": 0.6675245761871338, "reward_meter_mean": 0.7549527883529663, "reward_meter_std": 0.26335862278938293, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9893888235092163, "reward_repeat_soft_std": 0.007539966143667698, "reward_judge_quality_mean": 0.6737500429153442, "reward_judge_quality_std": 0.263435423374176, "reward_total_composite_mean": 0.6675245761871338, "reward_total_composite_std": 0.15883620083332062} {"timestamp_utc": "2026-04-13T08:23:59Z", "mode": "train", "global_step": 416, "epoch": 0.04178804620793571, "loss": 0.0218, "grad_norm": 13.643056869506836, "learning_rate": 8.742424242424243e-06, "num_tokens": 753220.0, "completions/mean_length": 40.5, "completions/min_length": 37.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.5, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.9194313287734985, "rewards/meter/std": 0.13804028928279877, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9871430993080139, "rewards/repeat_soft/std": 0.01945396140217781, "rewards/judge_quality/mean": 0.6762500405311584, "rewards/judge_quality/std": 0.2018442004919052, "rewards/total_composite/mean": 0.7491375207901001, "rewards/total_composite/std": 0.13408410549163818, "reward": 0.7491375207901001, "reward_std": 0.13408410549163818, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17899270355701447, "sampling/sampling_logp_difference/max": 1.3610343933105469, "sampling/importance_sampling_ratio/min": 0.38020259141921997, "sampling/importance_sampling_ratio/mean": 1.0316815376281738, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6855877041816711, "clip_ratio/low_mean": 0.09274809807538986, "clip_ratio/low_min": 0.09274809807538986, "clip_ratio/high_mean": 0.08824772294610739, "clip_ratio/high_max": 0.08824772294610739, "clip_ratio/region_mean": 0.18099582102149725, "reward_total_mean": 0.7491375207901001, "reward_meter_mean": 0.9194313287734985, "reward_meter_std": 0.13804028928279877, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9871430993080139, "reward_repeat_soft_std": 0.01945396140217781, "reward_judge_quality_mean": 0.6762500405311584, "reward_judge_quality_std": 0.2018442004919052, "reward_total_composite_mean": 0.7491375207901001, "reward_total_composite_std": 0.13408410549163818} {"timestamp_utc": "2026-04-13T08:24:06Z", "mode": "train", "global_step": 417, "epoch": 0.0418884982420894, "loss": 0.0458, "grad_norm": 9.444697380065918, "learning_rate": 8.73939393939394e-06, "num_tokens": 755195.0, "completions/mean_length": 70.875, "completions/min_length": 61.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.875, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9849028587341309, "rewards/meter/std": 0.014686250127851963, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8227899074554443, "rewards/repeat_soft/std": 0.1468363106250763, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.12906256318092346, "rewards/total_composite/mean": 0.6057982444763184, "rewards/total_composite/std": 0.09362467378377914, "reward": 0.6057982444763184, "reward_std": 0.09362467378377914, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14614373445510864, "sampling/sampling_logp_difference/max": 1.246006727218628, "sampling/importance_sampling_ratio/min": 0.2876511812210083, "sampling/importance_sampling_ratio/mean": 1.0279991626739502, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1420061439275742, "clip_ratio/low_mean": 0.07545688562095165, "clip_ratio/low_min": 0.07545688562095165, "clip_ratio/high_mean": 0.06647481210529804, "clip_ratio/high_max": 0.06647481210529804, "clip_ratio/region_mean": 0.1419316977262497, "reward_total_mean": 0.6057982444763184, "reward_meter_mean": 0.9849028587341309, "reward_meter_std": 0.014686250127851963, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8227899074554443, "reward_repeat_soft_std": 0.1468363106250763, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.12906256318092346, "reward_total_composite_mean": 0.6057982444763184, "reward_total_composite_std": 0.09362467378377914} {"timestamp_utc": "2026-04-13T08:24:13Z", "mode": "train", "global_step": 418, "epoch": 0.041988950276243095, "loss": 0.0944, "grad_norm": 20.964733123779297, "learning_rate": 8.736363636363638e-06, "num_tokens": 756763.0, "completions/mean_length": 36.0, "completions/min_length": 26.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.0, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.7766556739807129, "rewards/meter/std": 0.33451172709465027, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9609910845756531, "rewards/repeat_soft/std": 0.045096565037965775, "rewards/judge_quality/mean": 0.5562499761581421, "rewards/judge_quality/std": 0.2249404937028885, "rewards/total_composite/mean": 0.6375292539596558, "rewards/total_composite/std": 0.19607028365135193, "reward": 0.6375292539596558, "reward_std": 0.19607029855251312, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2082994431257248, "sampling/sampling_logp_difference/max": 4.037148475646973, "sampling/importance_sampling_ratio/min": 0.017647724598646164, "sampling/importance_sampling_ratio/mean": 1.0093640089035034, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7629767209291458, "clip_ratio/low_mean": 0.1277181648183614, "clip_ratio/low_min": 0.1277181648183614, "clip_ratio/high_mean": 0.05000000074505806, "clip_ratio/high_max": 0.05000000074505806, "clip_ratio/region_mean": 0.17771816556341946, "reward_total_mean": 0.6375292539596558, "reward_meter_mean": 0.7766556739807129, "reward_meter_std": 0.33451172709465027, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9609910845756531, "reward_repeat_soft_std": 0.045096565037965775, "reward_judge_quality_mean": 0.5562499761581421, "reward_judge_quality_std": 0.2249404937028885, "reward_total_composite_mean": 0.6375292539596558, "reward_total_composite_std": 0.19607028365135193} {"timestamp_utc": "2026-04-13T08:24:21Z", "mode": "train", "global_step": 419, "epoch": 0.04208940231039678, "loss": 0.0766, "grad_norm": 7.3235578536987305, "learning_rate": 8.733333333333333e-06, "num_tokens": 759007.0, "completions/mean_length": 91.5, "completions/min_length": 75.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.5, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.9800565838813782, "rewards/meter/std": 0.016581416130065918, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9972222447395325, "rewards/repeat_soft/std": 0.0028130800928920507, "rewards/judge_quality/mean": 0.5399999618530273, "rewards/judge_quality/std": 0.22123032808303833, "rewards/total_composite/mean": 0.6941713094711304, "rewards/total_composite/std": 0.1425619125366211, "reward": 0.6941713094711304, "reward_std": 0.1425618976354599, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1567026972770691, "sampling/sampling_logp_difference/max": 1.4153804779052734, "sampling/importance_sampling_ratio/min": 0.242833212018013, "sampling/importance_sampling_ratio/mean": 1.029744029045105, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6311266124248505, "clip_ratio/low_mean": 0.06705222418531775, "clip_ratio/low_min": 0.06705222418531775, "clip_ratio/high_mean": 0.060451144352555275, "clip_ratio/high_max": 0.060451144352555275, "clip_ratio/region_mean": 0.12750336853787303, "reward_total_mean": 0.6941713094711304, "reward_meter_mean": 0.9800565838813782, "reward_meter_std": 0.016581416130065918, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9972222447395325, "reward_repeat_soft_std": 0.0028130800928920507, "reward_judge_quality_mean": 0.5399999618530273, "reward_judge_quality_std": 0.22123032808303833, "reward_total_composite_mean": 0.6941713094711304, "reward_total_composite_std": 0.1425619125366211} {"timestamp_utc": "2026-04-13T08:24:27Z", "mode": "train", "global_step": 420, "epoch": 0.04218985434455048, "loss": 0.1067, "grad_norm": 14.094193458557129, "learning_rate": 8.73030303030303e-06, "num_tokens": 760555.0, "completions/mean_length": 41.5, "completions/min_length": 30.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.5, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.31071364879608154, "rewards/meter/std": 0.2925502359867096, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9957492351531982, "rewards/repeat_soft/std": 0.005208555143326521, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.43526244163513184, "rewards/total_composite/std": 0.07953180372714996, "reward": 0.43526244163513184, "reward_std": 0.07953178882598877, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21063420176506042, "sampling/sampling_logp_difference/max": 1.7707998752593994, "sampling/importance_sampling_ratio/min": 0.1701968014240265, "sampling/importance_sampling_ratio/mean": 1.0322872400283813, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9791297540068626, "clip_ratio/low_mean": 0.13185427524149418, "clip_ratio/low_min": 0.13185427524149418, "clip_ratio/high_mean": 0.05131579004228115, "clip_ratio/high_max": 0.05131579004228115, "clip_ratio/region_mean": 0.18317006528377533, "reward_total_mean": 0.43526244163513184, "reward_meter_mean": 0.31071364879608154, "reward_meter_std": 0.2925502359867096, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9957492351531982, "reward_repeat_soft_std": 0.005208555143326521, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.43526244163513184, "reward_total_composite_std": 0.07953180372714996} {"timestamp_utc": "2026-04-13T08:24:35Z", "mode": "train", "global_step": 421, "epoch": 0.04229030637870417, "loss": 0.031, "grad_norm": 28.130706787109375, "learning_rate": 8.727272727272728e-06, "num_tokens": 762151.0, "completions/mean_length": 22.5, "completions/min_length": 18.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.5, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.8638845086097717, "rewards/meter/std": 0.3477573096752167, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.5874203443527222, "rewards/total_composite/std": 0.09827587753534317, "reward": 0.5874203443527222, "reward_std": 0.09827586263418198, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18736852705478668, "sampling/sampling_logp_difference/max": 2.2155232429504395, "sampling/importance_sampling_ratio/min": 0.10909641534090042, "sampling/importance_sampling_ratio/mean": 1.0350109338760376, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.808619111776352, "clip_ratio/low_mean": 0.01785714365541935, "clip_ratio/low_min": 0.01785714365541935, "clip_ratio/high_mean": 0.1561513552442193, "clip_ratio/high_max": 0.1561513552442193, "clip_ratio/region_mean": 0.17400849889963865, "reward_total_mean": 0.5874203443527222, "reward_meter_mean": 0.8638845086097717, "reward_meter_std": 0.3477573096752167, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.5874203443527222, "reward_total_composite_std": 0.09827587753534317} {"timestamp_utc": "2026-04-13T08:24:42Z", "mode": "train", "global_step": 422, "epoch": 0.04239075841285786, "loss": 0.0368, "grad_norm": 17.89279556274414, "learning_rate": 8.724242424242425e-06, "num_tokens": 763698.0, "completions/mean_length": 31.375, "completions/min_length": 27.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.375, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9931104183197021, "rewards/meter/std": 0.002878874307498336, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.942622184753418, "rewards/repeat_soft/std": 0.06343577057123184, "rewards/judge_quality/mean": 0.6737500429153442, "rewards/judge_quality/std": 0.263435423374176, "rewards/total_composite/mean": 0.6243409514427185, "rewards/total_composite/std": 0.4099300503730774, "reward": 0.6243409514427185, "reward_std": 0.409930020570755, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1619260013103485, "sampling/sampling_logp_difference/max": 1.5931119918823242, "sampling/importance_sampling_ratio/min": 0.20329199731349945, "sampling/importance_sampling_ratio/mean": 1.027131199836731, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.269327849149704, "clip_ratio/low_mean": 0.04277935717254877, "clip_ratio/low_min": 0.04277935717254877, "clip_ratio/high_mean": 0.08713672962039709, "clip_ratio/high_max": 0.08713672962039709, "clip_ratio/region_mean": 0.12991608679294586, "reward_total_mean": 0.6243409514427185, "reward_meter_mean": 0.9931104183197021, "reward_meter_std": 0.002878874307498336, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.942622184753418, "reward_repeat_soft_std": 0.06343577057123184, "reward_judge_quality_mean": 0.6737500429153442, "reward_judge_quality_std": 0.263435423374176, "reward_total_composite_mean": 0.6243409514427185, "reward_total_composite_std": 0.4099300503730774} {"timestamp_utc": "2026-04-13T08:24:51Z", "mode": "train", "global_step": 423, "epoch": 0.042491210447011554, "loss": 0.0938, "grad_norm": 12.574085235595703, "learning_rate": 8.72121212121212e-06, "num_tokens": 765711.0, "completions/mean_length": 78.625, "completions/min_length": 51.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.625, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.39449363946914673, "rewards/meter/std": 0.24479669332504272, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9084656834602356, "rewards/repeat_soft/std": 0.2384941577911377, "rewards/judge_quality/mean": 0.5112500190734863, "rewards/judge_quality/std": 0.18216457962989807, "rewards/total_composite/mean": 0.4602924585342407, "rewards/total_composite/std": 0.09776712954044342, "reward": 0.4602924585342407, "reward_std": 0.09776712208986282, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19461190700531006, "sampling/sampling_logp_difference/max": 2.4785842895507812, "sampling/importance_sampling_ratio/min": 0.08386186510324478, "sampling/importance_sampling_ratio/mean": 1.010611891746521, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4617060646414757, "clip_ratio/low_mean": 0.07937775831669569, "clip_ratio/low_min": 0.07937775831669569, "clip_ratio/high_mean": 0.09583958983421326, "clip_ratio/high_max": 0.09583958983421326, "clip_ratio/region_mean": 0.17521734815090895, "reward_total_mean": 0.4602924585342407, "reward_meter_mean": 0.39449363946914673, "reward_meter_std": 0.24479669332504272, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9084656834602356, "reward_repeat_soft_std": 0.2384941577911377, "reward_judge_quality_mean": 0.5112500190734863, "reward_judge_quality_std": 0.18216457962989807, "reward_total_composite_mean": 0.4602924585342407, "reward_total_composite_std": 0.09776712954044342} {"timestamp_utc": "2026-04-13T08:25:00Z", "mode": "train", "global_step": 424, "epoch": 0.04259166248116524, "loss": 0.0904, "grad_norm": 14.377594947814941, "learning_rate": 8.71818181818182e-06, "num_tokens": 767528.0, "completions/mean_length": 56.125, "completions/min_length": 47.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.125, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.8052804470062256, "rewards/meter/std": 0.29650941491127014, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9423341751098633, "rewards/repeat_soft/std": 0.08976280689239502, "rewards/judge_quality/mean": 0.6449999809265137, "rewards/judge_quality/std": 0.24928471446037292, "rewards/total_composite/mean": 0.6919419169425964, "rewards/total_composite/std": 0.19053557515144348, "reward": 0.6919419169425964, "reward_std": 0.1905355602502823, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20590421557426453, "sampling/sampling_logp_difference/max": 1.8177146911621094, "sampling/importance_sampling_ratio/min": 0.16239644587039948, "sampling/importance_sampling_ratio/mean": 1.0273287296295166, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3588280379772186, "clip_ratio/low_mean": 0.08734078239649534, "clip_ratio/low_min": 0.08734078239649534, "clip_ratio/high_mean": 0.10055535100400448, "clip_ratio/high_max": 0.10055535100400448, "clip_ratio/region_mean": 0.18789613340049982, "reward_total_mean": 0.6919419169425964, "reward_meter_mean": 0.8052804470062256, "reward_meter_std": 0.29650941491127014, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9423341751098633, "reward_repeat_soft_std": 0.08976280689239502, "reward_judge_quality_mean": 0.6449999809265137, "reward_judge_quality_std": 0.24928471446037292, "reward_total_composite_mean": 0.6919419169425964, "reward_total_composite_std": 0.19053557515144348} {"timestamp_utc": "2026-04-13T08:25:07Z", "mode": "train", "global_step": 425, "epoch": 0.04269211451531894, "loss": 0.1283, "grad_norm": 24.715789794921875, "learning_rate": 8.715151515151515e-06, "num_tokens": 769222.0, "completions/mean_length": 30.75, "completions/min_length": 28.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.75, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.8295538425445557, "rewards/meter/std": 0.3294765055179596, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9836940765380859, "rewards/repeat_soft/std": 0.010605689138174057, "rewards/judge_quality/mean": 0.8612500429153442, "rewards/judge_quality/std": 0.16617010533809662, "rewards/total_composite/mean": 0.8057328462600708, "rewards/total_composite/std": 0.20401300489902496, "reward": 0.8057328462600708, "reward_std": 0.20401298999786377, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12585145235061646, "sampling/sampling_logp_difference/max": 1.3707711696624756, "sampling/importance_sampling_ratio/min": 0.2539110481739044, "sampling/importance_sampling_ratio/mean": 0.9612415432929993, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.530036062002182, "clip_ratio/low_mean": 0.030766253359615803, "clip_ratio/low_min": 0.030766253359615803, "clip_ratio/high_mean": 0.05957115115597844, "clip_ratio/high_max": 0.05957115115597844, "clip_ratio/region_mean": 0.09033740451559424, "reward_total_mean": 0.8057328462600708, "reward_meter_mean": 0.8295538425445557, "reward_meter_std": 0.3294765055179596, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9836940765380859, "reward_repeat_soft_std": 0.010605689138174057, "reward_judge_quality_mean": 0.8612500429153442, "reward_judge_quality_std": 0.16617010533809662, "reward_total_composite_mean": 0.8057328462600708, "reward_total_composite_std": 0.20401300489902496} {"timestamp_utc": "2026-04-13T08:25:15Z", "mode": "train", "global_step": 426, "epoch": 0.042792566549472624, "loss": 0.0414, "grad_norm": 15.755805015563965, "learning_rate": 8.712121212121212e-06, "num_tokens": 770914.0, "completions/mean_length": 49.5, "completions/min_length": 41.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.5, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.7202902436256409, "rewards/meter/std": 0.3197583556175232, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9715853929519653, "rewards/repeat_soft/std": 0.025107305496931076, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.22038927674293518, "rewards/total_composite/mean": 0.6496169567108154, "rewards/total_composite/std": 0.1708659678697586, "reward": 0.6496169567108154, "reward_std": 0.1708659678697586, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2259758710861206, "sampling/sampling_logp_difference/max": 1.8252153396606445, "sampling/importance_sampling_ratio/min": 0.1611829400062561, "sampling/importance_sampling_ratio/mean": 1.0428838729858398, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.802308902144432, "clip_ratio/low_mean": 0.10848477762192488, "clip_ratio/low_min": 0.10848477762192488, "clip_ratio/high_mean": 0.06292749661952257, "clip_ratio/high_max": 0.06292749661952257, "clip_ratio/region_mean": 0.17141227424144745, "reward_total_mean": 0.6496169567108154, "reward_meter_mean": 0.7202902436256409, "reward_meter_std": 0.3197583556175232, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9715853929519653, "reward_repeat_soft_std": 0.025107305496931076, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.22038927674293518, "reward_total_composite_mean": 0.6496169567108154, "reward_total_composite_std": 0.1708659678697586} {"timestamp_utc": "2026-04-13T08:25:24Z", "mode": "train", "global_step": 427, "epoch": 0.04289301858362632, "loss": -0.0635, "grad_norm": 8.554418563842773, "learning_rate": 8.70909090909091e-06, "num_tokens": 773425.0, "completions/mean_length": 132.875, "completions/min_length": 84.0, "completions/max_length": 219.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 132.875, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 219.0, "rewards/meter/mean": 0.9280627965927124, "rewards/meter/std": 0.16058552265167236, "rewards/count_adherence/mean": 0.7916666269302368, "rewards/count_adherence/std": 0.11785111576318741, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9765769839286804, "rewards/repeat_soft/std": 0.025364678353071213, "rewards/judge_quality/mean": 0.5824999809265137, "rewards/judge_quality/std": 0.176776722073555, "rewards/total_composite/mean": 0.505779504776001, "rewards/total_composite/std": 0.3327721655368805, "reward": 0.505779504776001, "reward_std": 0.3327721655368805, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17246946692466736, "sampling/sampling_logp_difference/max": 1.9859402179718018, "sampling/importance_sampling_ratio/min": 0.13725151121616364, "sampling/importance_sampling_ratio/mean": 1.025324821472168, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.558809533715248, "clip_ratio/low_mean": 0.04445033520460129, "clip_ratio/low_min": 0.04445033520460129, "clip_ratio/high_mean": 0.1237593786790967, "clip_ratio/high_max": 0.1237593786790967, "clip_ratio/region_mean": 0.168209713883698, "reward_total_mean": 0.505779504776001, "reward_meter_mean": 0.9280627965927124, "reward_meter_std": 0.16058552265167236, "reward_count_adherence_mean": 0.7916666269302368, "reward_count_adherence_std": 0.11785111576318741, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9765769839286804, "reward_repeat_soft_std": 0.025364678353071213, "reward_judge_quality_mean": 0.5824999809265137, "reward_judge_quality_std": 0.176776722073555, "reward_total_composite_mean": 0.505779504776001, "reward_total_composite_std": 0.3327721655368805} {"timestamp_utc": "2026-04-13T08:25:32Z", "mode": "train", "global_step": 428, "epoch": 0.04299347061778001, "loss": 0.0752, "grad_norm": 17.699542999267578, "learning_rate": 8.706060606060607e-06, "num_tokens": 775099.0, "completions/mean_length": 42.25, "completions/min_length": 32.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.25, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.4861021041870117, "rewards/meter/std": 0.359035462141037, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9769493937492371, "rewards/repeat_soft/std": 0.029769068583846092, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.4864930510520935, "rewards/total_composite/std": 0.1076832190155983, "reward": 0.4864930510520935, "reward_std": 0.1076832041144371, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2017541229724884, "sampling/sampling_logp_difference/max": 1.6000213623046875, "sampling/importance_sampling_ratio/min": 0.2018921971321106, "sampling/importance_sampling_ratio/mean": 1.0370087623596191, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.544850543141365, "clip_ratio/low_mean": 0.12223918549716473, "clip_ratio/low_min": 0.12223918549716473, "clip_ratio/high_mean": 0.057329047471284866, "clip_ratio/high_max": 0.057329047471284866, "clip_ratio/region_mean": 0.1795682329684496, "reward_total_mean": 0.4864930510520935, "reward_meter_mean": 0.4861021041870117, "reward_meter_std": 0.359035462141037, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9769493937492371, "reward_repeat_soft_std": 0.029769068583846092, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.4864930510520935, "reward_total_composite_std": 0.1076832190155983} {"timestamp_utc": "2026-04-13T08:25:41Z", "mode": "train", "global_step": 429, "epoch": 0.0430939226519337, "loss": 0.117, "grad_norm": 11.39891529083252, "learning_rate": 8.703030303030304e-06, "num_tokens": 777565.0, "completions/mean_length": 129.25, "completions/min_length": 106.0, "completions/max_length": 177.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 129.25, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 177.0, "rewards/meter/mean": 0.9440456628799438, "rewards/meter/std": 0.10624020546674728, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9840169548988342, "rewards/repeat_soft/std": 0.019269311800599098, "rewards/judge_quality/mean": 0.5699999928474426, "rewards/judge_quality/std": 0.16035676002502441, "rewards/total_composite/mean": 0.5813806653022766, "rewards/total_composite/std": 0.250386506319046, "reward": 0.5813806653022766, "reward_std": 0.250386506319046, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20710064470767975, "sampling/sampling_logp_difference/max": 2.949585437774658, "sampling/importance_sampling_ratio/min": 0.05236141011118889, "sampling/importance_sampling_ratio/mean": 1.0383763313293457, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0161682814359665, "clip_ratio/low_mean": 0.09372947178781033, "clip_ratio/low_min": 0.09372947178781033, "clip_ratio/high_mean": 0.08874286524951458, "clip_ratio/high_max": 0.08874286524951458, "clip_ratio/region_mean": 0.1824723370373249, "reward_total_mean": 0.5813806653022766, "reward_meter_mean": 0.9440456628799438, "reward_meter_std": 0.10624020546674728, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9840169548988342, "reward_repeat_soft_std": 0.019269311800599098, "reward_judge_quality_mean": 0.5699999928474426, "reward_judge_quality_std": 0.16035676002502441, "reward_total_composite_mean": 0.5813806653022766, "reward_total_composite_std": 0.250386506319046} {"timestamp_utc": "2026-04-13T08:25:47Z", "mode": "train", "global_step": 430, "epoch": 0.043194374686087396, "loss": 0.0519, "grad_norm": 15.2701416015625, "learning_rate": 8.700000000000001e-06, "num_tokens": 778826.0, "completions/mean_length": 26.625, "completions/min_length": 19.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.625, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.3605045676231384, "rewards/meter/std": 0.31919997930526733, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.8237500190734863, "rewards/judge_quality/std": 0.2722361385822296, "rewards/total_composite/mean": 0.5100744366645813, "rewards/total_composite/std": 0.16296890377998352, "reward": 0.5100744366645813, "reward_std": 0.16296890377998352, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18163342773914337, "sampling/sampling_logp_difference/max": 2.0839452743530273, "sampling/importance_sampling_ratio/min": 0.12443829327821732, "sampling/importance_sampling_ratio/mean": 1.0236661434173584, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6269883215427399, "clip_ratio/low_mean": 0.07251747930422425, "clip_ratio/low_min": 0.07251747930422425, "clip_ratio/high_mean": 0.07679007947444916, "clip_ratio/high_max": 0.07679007947444916, "clip_ratio/region_mean": 0.1493075587786734, "reward_total_mean": 0.5100744366645813, "reward_meter_mean": 0.3605045676231384, "reward_meter_std": 0.31919997930526733, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.8237500190734863, "reward_judge_quality_std": 0.2722361385822296, "reward_total_composite_mean": 0.5100744366645813, "reward_total_composite_std": 0.16296890377998352} {"timestamp_utc": "2026-04-13T08:25:55Z", "mode": "train", "global_step": 431, "epoch": 0.043294826720241084, "loss": -0.0001, "grad_norm": 17.1655330657959, "learning_rate": 8.696969696969699e-06, "num_tokens": 780305.0, "completions/mean_length": 36.875, "completions/min_length": 33.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.875, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.4635351300239563, "rewards/meter/std": 0.3749304711818695, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9966304302215576, "rewards/repeat_soft/std": 0.004131971392780542, "rewards/judge_quality/mean": 0.6675000190734863, "rewards/judge_quality/std": 0.1348809152841568, "rewards/total_composite/mean": 0.48579472303390503, "rewards/total_composite/std": 0.2634431719779968, "reward": 0.48579472303390503, "reward_std": 0.2634431719779968, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2026931643486023, "sampling/sampling_logp_difference/max": 1.8704214096069336, "sampling/importance_sampling_ratio/min": 0.15405872464179993, "sampling/importance_sampling_ratio/mean": 1.00552237033844, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.370562307536602, "clip_ratio/low_mean": 0.09522218257188797, "clip_ratio/low_min": 0.09522218257188797, "clip_ratio/high_mean": 0.07766646798700094, "clip_ratio/high_max": 0.07766646798700094, "clip_ratio/region_mean": 0.1728886505588889, "reward_total_mean": 0.48579472303390503, "reward_meter_mean": 0.4635351300239563, "reward_meter_std": 0.3749304711818695, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9966304302215576, "reward_repeat_soft_std": 0.004131971392780542, "reward_judge_quality_mean": 0.6675000190734863, "reward_judge_quality_std": 0.1348809152841568, "reward_total_composite_mean": 0.48579472303390503, "reward_total_composite_std": 0.2634431719779968} {"timestamp_utc": "2026-04-13T08:26:03Z", "mode": "train", "global_step": 432, "epoch": 0.04339527875439478, "loss": 0.0515, "grad_norm": 7.66887092590332, "learning_rate": 8.693939393939394e-06, "num_tokens": 783010.0, "completions/mean_length": 123.125, "completions/min_length": 110.0, "completions/max_length": 138.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 123.125, "completions/min_terminated_length": 110.0, "completions/max_terminated_length": 138.0, "rewards/meter/mean": 0.9879873991012573, "rewards/meter/std": 0.009452181868255138, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8556176424026489, "rewards/repeat_soft/std": 0.10616997629404068, "rewards/judge_quality/mean": 0.5325000286102295, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.5374159216880798, "rewards/total_composite/std": 0.23432575166225433, "reward": 0.5374159216880798, "reward_std": 0.23432573676109314, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16589200496673584, "sampling/sampling_logp_difference/max": 1.4953727722167969, "sampling/importance_sampling_ratio/min": 0.22416503727436066, "sampling/importance_sampling_ratio/mean": 1.0411186218261719, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5380447953939438, "clip_ratio/low_mean": 0.028368454426527023, "clip_ratio/low_min": 0.028368454426527023, "clip_ratio/high_mean": 0.1173196341842413, "clip_ratio/high_max": 0.1173196341842413, "clip_ratio/region_mean": 0.14568808861076832, "reward_total_mean": 0.5374159216880798, "reward_meter_mean": 0.9879873991012573, "reward_meter_std": 0.009452181868255138, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8556176424026489, "reward_repeat_soft_std": 0.10616997629404068, "reward_judge_quality_mean": 0.5325000286102295, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.5374159216880798, "reward_total_composite_std": 0.23432575166225433} {"timestamp_utc": "2026-04-13T08:26:09Z", "mode": "train", "global_step": 433, "epoch": 0.043495730788548466, "loss": 0.0247, "grad_norm": 17.924205780029297, "learning_rate": 8.690909090909091e-06, "num_tokens": 784669.0, "completions/mean_length": 36.375, "completions/min_length": 31.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.375, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.9752165079116821, "rewards/meter/std": 0.040312111377716064, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9120166897773743, "rewards/repeat_soft/std": 0.10026796162128448, "rewards/judge_quality/mean": 0.5562499761581421, "rewards/judge_quality/std": 0.2249404937028885, "rewards/total_composite/mean": 0.6149869561195374, "rewards/total_composite/std": 0.2904825508594513, "reward": 0.6149869561195374, "reward_std": 0.2904825508594513, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1944267749786377, "sampling/sampling_logp_difference/max": 1.400411605834961, "sampling/importance_sampling_ratio/min": 0.24649548530578613, "sampling/importance_sampling_ratio/mean": 1.0463149547576904, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9324277490377426, "clip_ratio/low_mean": 0.08154762163758278, "clip_ratio/low_min": 0.08154762163758278, "clip_ratio/high_mean": 0.08843881450593472, "clip_ratio/high_max": 0.08843881450593472, "clip_ratio/region_mean": 0.1699864361435175, "reward_total_mean": 0.6149869561195374, "reward_meter_mean": 0.9752165079116821, "reward_meter_std": 0.040312111377716064, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9120166897773743, "reward_repeat_soft_std": 0.10026796162128448, "reward_judge_quality_mean": 0.5562499761581421, "reward_judge_quality_std": 0.2249404937028885, "reward_total_composite_mean": 0.6149869561195374, "reward_total_composite_std": 0.2904825508594513} {"timestamp_utc": "2026-04-13T08:26:16Z", "mode": "train", "global_step": 434, "epoch": 0.04359618282270216, "loss": -0.2025, "grad_norm": 7.984838008880615, "learning_rate": 8.687878787878789e-06, "num_tokens": 786349.0, "completions/mean_length": 56.0, "completions/min_length": 17.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.0, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.7314310073852539, "rewards/meter/std": 0.2750467360019684, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.2357022762298584, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9881883859634399, "rewards/repeat_soft/std": 0.013463444076478481, "rewards/judge_quality/mean": 0.5062500238418579, "rewards/judge_quality/std": 0.2018442004919052, "rewards/total_composite/mean": 0.5222692489624023, "rewards/total_composite/std": 0.24057649075984955, "reward": 0.5222692489624023, "reward_std": 0.24057647585868835, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1608295440673828, "sampling/sampling_logp_difference/max": 3.9674124717712402, "sampling/importance_sampling_ratio/min": 0.018922332674264908, "sampling/importance_sampling_ratio/mean": 1.0018035173416138, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7854427397251129, "clip_ratio/low_mean": 0.05656746681779623, "clip_ratio/low_min": 0.05656746681779623, "clip_ratio/high_mean": 0.10639818571507931, "clip_ratio/high_max": 0.10639818571507931, "clip_ratio/region_mean": 0.16296565253287554, "reward_total_mean": 0.5222692489624023, "reward_meter_mean": 0.7314310073852539, "reward_meter_std": 0.2750467360019684, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.2357022762298584, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9881883859634399, "reward_repeat_soft_std": 0.013463444076478481, "reward_judge_quality_mean": 0.5062500238418579, "reward_judge_quality_std": 0.2018442004919052, "reward_total_composite_mean": 0.5222692489624023, "reward_total_composite_std": 0.24057649075984955} {"timestamp_utc": "2026-04-13T08:26:22Z", "mode": "train", "global_step": 435, "epoch": 0.04369663485685585, "loss": 0.0547, "grad_norm": 30.92169189453125, "learning_rate": 8.684848484848486e-06, "num_tokens": 787744.0, "completions/mean_length": 17.375, "completions/min_length": 15.0, "completions/max_length": 20.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 17.375, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 20.0, "rewards/meter/mean": 0.7917190194129944, "rewards/meter/std": 0.27918416261672974, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9305113554000854, "rewards/repeat_soft/std": 0.07009010016918182, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.0645727664232254, "rewards/total_composite/mean": 0.5465813279151917, "rewards/total_composite/std": 0.08320525288581848, "reward": 0.5465813279151917, "reward_std": 0.08320525288581848, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1961645632982254, "sampling/sampling_logp_difference/max": 1.2406105995178223, "sampling/importance_sampling_ratio/min": 0.2892075777053833, "sampling/importance_sampling_ratio/mean": 1.047686219215393, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.589640535414219, "clip_ratio/low_mean": 0.07426470797508955, "clip_ratio/low_min": 0.07426470797508955, "clip_ratio/high_mean": 0.10399305773898959, "clip_ratio/high_max": 0.10399305773898959, "clip_ratio/region_mean": 0.17825776571407914, "reward_total_mean": 0.5465813279151917, "reward_meter_mean": 0.7917190194129944, "reward_meter_std": 0.27918416261672974, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9305113554000854, "reward_repeat_soft_std": 0.07009010016918182, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.0645727664232254, "reward_total_composite_mean": 0.5465813279151917, "reward_total_composite_std": 0.08320525288581848} {"timestamp_utc": "2026-04-13T08:26:30Z", "mode": "train", "global_step": 436, "epoch": 0.04379708689100954, "loss": 0.0478, "grad_norm": 11.516942977905273, "learning_rate": 8.681818181818182e-06, "num_tokens": 789814.0, "completions/mean_length": 76.75, "completions/min_length": 61.0, "completions/max_length": 89.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.75, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 89.0, "rewards/meter/mean": 0.6057450771331787, "rewards/meter/std": 0.2812247574329376, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9964901804924011, "rewards/repeat_soft/std": 0.003044117009267211, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.55422043800354, "rewards/total_composite/std": 0.13206170499324799, "reward": 0.55422043800354, "reward_std": 0.13206170499324799, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1947910338640213, "sampling/sampling_logp_difference/max": 1.2177276611328125, "sampling/importance_sampling_ratio/min": 0.2959018051624298, "sampling/importance_sampling_ratio/mean": 1.017616629600525, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0375936776399612, "clip_ratio/low_mean": 0.09038877952843904, "clip_ratio/low_min": 0.09038877952843904, "clip_ratio/high_mean": 0.07839574478566647, "clip_ratio/high_max": 0.07839574478566647, "clip_ratio/region_mean": 0.1687845243141055, "reward_total_mean": 0.55422043800354, "reward_meter_mean": 0.6057450771331787, "reward_meter_std": 0.2812247574329376, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9964901804924011, "reward_repeat_soft_std": 0.003044117009267211, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.55422043800354, "reward_total_composite_std": 0.13206170499324799} {"timestamp_utc": "2026-04-13T08:26:38Z", "mode": "train", "global_step": 437, "epoch": 0.04389753892516324, "loss": 0.0099, "grad_norm": 13.20158863067627, "learning_rate": 8.67878787878788e-06, "num_tokens": 791820.0, "completions/mean_length": 64.75, "completions/min_length": 59.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.75, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9005851745605469, "rewards/meter/std": 0.21653418242931366, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9773539900779724, "rewards/repeat_soft/std": 0.027312718331813812, "rewards/judge_quality/mean": 0.5950000286102295, "rewards/judge_quality/std": 0.19820626080036163, "rewards/total_composite/mean": 0.6135097742080688, "rewards/total_composite/std": 0.2840389013290405, "reward": 0.6135097742080688, "reward_std": 0.2840389013290405, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20088978111743927, "sampling/sampling_logp_difference/max": 1.7408978939056396, "sampling/importance_sampling_ratio/min": 0.17536288499832153, "sampling/importance_sampling_ratio/mean": 1.0183771848678589, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6622829139232635, "clip_ratio/low_mean": 0.1009948905557394, "clip_ratio/low_min": 0.1009948905557394, "clip_ratio/high_mean": 0.1130131408572197, "clip_ratio/high_max": 0.1130131408572197, "clip_ratio/region_mean": 0.2140080314129591, "reward_total_mean": 0.6135097742080688, "reward_meter_mean": 0.9005851745605469, "reward_meter_std": 0.21653418242931366, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9773539900779724, "reward_repeat_soft_std": 0.027312718331813812, "reward_judge_quality_mean": 0.5950000286102295, "reward_judge_quality_std": 0.19820626080036163, "reward_total_composite_mean": 0.6135097742080688, "reward_total_composite_std": 0.2840389013290405} {"timestamp_utc": "2026-04-13T08:26:45Z", "mode": "train", "global_step": 438, "epoch": 0.043997990959316925, "loss": -0.0558, "grad_norm": 15.111183166503906, "learning_rate": 8.675757575757576e-06, "num_tokens": 793404.0, "completions/mean_length": 39.0, "completions/min_length": 32.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.7226242423057556, "rewards/meter/std": 0.3787357211112976, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9969919919967651, "rewards/repeat_soft/std": 0.006377330049872398, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.13695022463798523, "rewards/total_composite/mean": 0.5948572158813477, "rewards/total_composite/std": 0.16125662624835968, "reward": 0.5948572158813477, "reward_std": 0.16125662624835968, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19764646887779236, "sampling/sampling_logp_difference/max": 1.359586238861084, "sampling/importance_sampling_ratio/min": 0.25676700472831726, "sampling/importance_sampling_ratio/mean": 1.0119014978408813, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7450518012046814, "clip_ratio/low_mean": 0.05391471879556775, "clip_ratio/low_min": 0.05391471879556775, "clip_ratio/high_mean": 0.1409989334642887, "clip_ratio/high_max": 0.1409989334642887, "clip_ratio/region_mean": 0.19491365225985646, "reward_total_mean": 0.5948572158813477, "reward_meter_mean": 0.7226242423057556, "reward_meter_std": 0.3787357211112976, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9969919919967651, "reward_repeat_soft_std": 0.006377330049872398, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.13695022463798523, "reward_total_composite_mean": 0.5948572158813477, "reward_total_composite_std": 0.16125662624835968} {"timestamp_utc": "2026-04-13T08:26:52Z", "mode": "train", "global_step": 439, "epoch": 0.04409844299347062, "loss": -0.0189, "grad_norm": 16.915756225585938, "learning_rate": 8.672727272727273e-06, "num_tokens": 794921.0, "completions/mean_length": 37.625, "completions/min_length": 33.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.625, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.5057054758071899, "rewards/meter/std": 0.4522855579853058, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9985843300819397, "rewards/repeat_soft/std": 0.0036601137835532427, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.47689583897590637, "rewards/total_composite/std": 0.26841363310813904, "reward": 0.47689583897590637, "reward_std": 0.26841360330581665, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21142157912254333, "sampling/sampling_logp_difference/max": 1.6103746891021729, "sampling/importance_sampling_ratio/min": 0.36710667610168457, "sampling/importance_sampling_ratio/mean": 1.0207772254943848, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.390752464532852, "clip_ratio/low_mean": 0.0848891967907548, "clip_ratio/low_min": 0.0848891967907548, "clip_ratio/high_mean": 0.11912698484957218, "clip_ratio/high_max": 0.11912698484957218, "clip_ratio/region_mean": 0.20401618164032698, "reward_total_mean": 0.47689583897590637, "reward_meter_mean": 0.5057054758071899, "reward_meter_std": 0.4522855579853058, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9985843300819397, "reward_repeat_soft_std": 0.0036601137835532427, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.47689583897590637, "reward_total_composite_std": 0.26841363310813904} {"timestamp_utc": "2026-04-13T08:27:00Z", "mode": "train", "global_step": 440, "epoch": 0.04419889502762431, "loss": 0.0282, "grad_norm": 10.141862869262695, "learning_rate": 8.66969696969697e-06, "num_tokens": 797454.0, "completions/mean_length": 103.625, "completions/min_length": 84.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 103.625, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.6514586806297302, "rewards/meter/std": 0.2366323471069336, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.1035098284482956, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9928970336914062, "rewards/repeat_soft/std": 0.004234022926539183, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.5241789817810059, "rewards/total_composite/std": 0.11570888757705688, "reward": 0.5241789817810059, "reward_std": 0.11570888012647629, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21444502472877502, "sampling/sampling_logp_difference/max": 1.3234453201293945, "sampling/importance_sampling_ratio/min": 0.266216516494751, "sampling/importance_sampling_ratio/mean": 1.0347667932510376, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.9164059162139893, "clip_ratio/low_mean": 0.10031036101281643, "clip_ratio/low_min": 0.10031036101281643, "clip_ratio/high_mean": 0.07623598724603653, "clip_ratio/high_max": 0.07623598724603653, "clip_ratio/region_mean": 0.17654634825885296, "reward_total_mean": 0.5241789817810059, "reward_meter_mean": 0.6514586806297302, "reward_meter_std": 0.2366323471069336, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.1035098284482956, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9928970336914062, "reward_repeat_soft_std": 0.004234022926539183, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.5241789817810059, "reward_total_composite_std": 0.11570888757705688} {"timestamp_utc": "2026-04-13T08:27:07Z", "mode": "train", "global_step": 441, "epoch": 0.044299347061778, "loss": 0.0436, "grad_norm": 24.817583084106445, "learning_rate": 8.666666666666668e-06, "num_tokens": 798797.0, "completions/mean_length": 28.875, "completions/min_length": 25.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.875, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.5134502649307251, "rewards/meter/std": 0.39729589223861694, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9864075183868408, "rewards/repeat_soft/std": 0.013196037150919437, "rewards/judge_quality/mean": 0.6487500071525574, "rewards/judge_quality/std": 0.2507951855659485, "rewards/total_composite/mean": 0.572398841381073, "rewards/total_composite/std": 0.21716076135635376, "reward": 0.572398841381073, "reward_std": 0.21716077625751495, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17735087871551514, "sampling/sampling_logp_difference/max": 1.9800612926483154, "sampling/importance_sampling_ratio/min": 0.1540224403142929, "sampling/importance_sampling_ratio/mean": 0.9859204292297363, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0950158163905144, "clip_ratio/low_mean": 0.08132441435009241, "clip_ratio/low_min": 0.08132441435009241, "clip_ratio/high_mean": 0.08135775849223137, "clip_ratio/high_max": 0.08135775849223137, "clip_ratio/region_mean": 0.16268217284232378, "reward_total_mean": 0.572398841381073, "reward_meter_mean": 0.5134502649307251, "reward_meter_std": 0.39729589223861694, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9864075183868408, "reward_repeat_soft_std": 0.013196037150919437, "reward_judge_quality_mean": 0.6487500071525574, "reward_judge_quality_std": 0.2507951855659485, "reward_total_composite_mean": 0.572398841381073, "reward_total_composite_std": 0.21716076135635376} {"timestamp_utc": "2026-04-13T08:27:13Z", "mode": "train", "global_step": 442, "epoch": 0.04439979909593169, "loss": -0.0043, "grad_norm": 15.964285850524902, "learning_rate": 8.663636363636363e-06, "num_tokens": 800289.0, "completions/mean_length": 34.5, "completions/min_length": 30.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.5, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.7339317798614502, "rewards/meter/std": 0.3718337118625641, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9568861722946167, "rewards/repeat_soft/std": 0.06399570405483246, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.5337520837783813, "rewards/total_composite/std": 0.1071075052022934, "reward": 0.5337520837783813, "reward_std": 0.1071075052022934, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24122373759746552, "sampling/sampling_logp_difference/max": 1.3552074432373047, "sampling/importance_sampling_ratio/min": 0.25789380073547363, "sampling/importance_sampling_ratio/mean": 1.0309560298919678, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.323012739419937, "clip_ratio/low_mean": 0.07353896275162697, "clip_ratio/low_min": 0.07353896275162697, "clip_ratio/high_mean": 0.10821158112958074, "clip_ratio/high_max": 0.10821158112958074, "clip_ratio/region_mean": 0.1817505438812077, "reward_total_mean": 0.5337520837783813, "reward_meter_mean": 0.7339317798614502, "reward_meter_std": 0.3718337118625641, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9568861722946167, "reward_repeat_soft_std": 0.06399570405483246, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.5337520837783813, "reward_total_composite_std": 0.1071075052022934} {"timestamp_utc": "2026-04-13T08:27:22Z", "mode": "train", "global_step": 443, "epoch": 0.044500251130085385, "loss": 0.2345, "grad_norm": 11.721576690673828, "learning_rate": 8.660606060606062e-06, "num_tokens": 802296.0, "completions/mean_length": 74.875, "completions/min_length": 46.0, "completions/max_length": 157.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.875, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 157.0, "rewards/meter/mean": 0.5554511547088623, "rewards/meter/std": 0.35469502210617065, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9882750511169434, "rewards/repeat_soft/std": 0.011020518839359283, "rewards/judge_quality/mean": 0.3799999952316284, "rewards/judge_quality/std": 0.1256980448961258, "rewards/total_composite/mean": 0.4977571368217468, "rewards/total_composite/std": 0.10918775945901871, "reward": 0.4977571368217468, "reward_std": 0.10918774455785751, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23504464328289032, "sampling/sampling_logp_difference/max": 1.4760236740112305, "sampling/importance_sampling_ratio/min": 0.2285446673631668, "sampling/importance_sampling_ratio/mean": 1.017876386642456, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.2571473568677902, "clip_ratio/low_mean": 0.041979849338531494, "clip_ratio/low_min": 0.041979849338531494, "clip_ratio/high_mean": 0.13672222942113876, "clip_ratio/high_max": 0.13672222942113876, "clip_ratio/region_mean": 0.17870207875967026, "reward_total_mean": 0.4977571368217468, "reward_meter_mean": 0.5554511547088623, "reward_meter_std": 0.35469502210617065, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9882750511169434, "reward_repeat_soft_std": 0.011020518839359283, "reward_judge_quality_mean": 0.3799999952316284, "reward_judge_quality_std": 0.1256980448961258, "reward_total_composite_mean": 0.4977571368217468, "reward_total_composite_std": 0.10918775945901871} {"timestamp_utc": "2026-04-13T08:27:34Z", "mode": "train", "global_step": 444, "epoch": 0.04460070316423908, "loss": -0.1005, "grad_norm": 4.826160907745361, "learning_rate": 8.657575757575758e-06, "num_tokens": 804366.0, "completions/mean_length": 132.75, "completions/min_length": 69.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 78.5714340209961, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.8388128280639648, "rewards/meter/std": 0.30288782715797424, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.1035098284482956, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.962822437286377, "rewards/repeat_soft/std": 0.02336599864065647, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.13593590259552002, "rewards/total_composite/mean": 0.4479786157608032, "rewards/total_composite/std": 0.27719706296920776, "reward": 0.4479786157608032, "reward_std": 0.27719706296920776, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21294830739498138, "sampling/sampling_logp_difference/max": 1.9244985580444336, "sampling/importance_sampling_ratio/min": 0.14594891667366028, "sampling/importance_sampling_ratio/mean": 1.0390417575836182, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0412132292985916, "clip_ratio/low_mean": 0.023148147389292717, "clip_ratio/low_min": 0.023148147389292717, "clip_ratio/high_mean": 0.15710760466754436, "clip_ratio/high_max": 0.15710760466754436, "clip_ratio/region_mean": 0.18025575205683708, "reward_total_mean": 0.4479786157608032, "reward_meter_mean": 0.8388128280639648, "reward_meter_std": 0.30288782715797424, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.1035098284482956, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.962822437286377, "reward_repeat_soft_std": 0.02336599864065647, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.13593590259552002, "reward_total_composite_mean": 0.4479786157608032, "reward_total_composite_std": 0.27719706296920776} {"timestamp_utc": "2026-04-13T08:27:42Z", "mode": "train", "global_step": 445, "epoch": 0.04470115519839277, "loss": 0.0424, "grad_norm": 8.42081069946289, "learning_rate": 8.654545454545455e-06, "num_tokens": 806720.0, "completions/mean_length": 111.25, "completions/min_length": 98.0, "completions/max_length": 125.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 111.25, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.9921140670776367, "rewards/meter/std": 0.005870350170880556, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8212906718254089, "rewards/repeat_soft/std": 0.18632890284061432, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.1524970829486847, "rewards/total_composite/mean": 0.5564401149749756, "rewards/total_composite/std": 0.10963989049196243, "reward": 0.5564401149749756, "reward_std": 0.10963989049196243, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1865592896938324, "sampling/sampling_logp_difference/max": 1.4757966995239258, "sampling/importance_sampling_ratio/min": 0.2285965532064438, "sampling/importance_sampling_ratio/mean": 1.0599757432937622, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.430830806493759, "clip_ratio/low_mean": 0.050699902698397636, "clip_ratio/low_min": 0.050699902698397636, "clip_ratio/high_mean": 0.09075277671217918, "clip_ratio/high_max": 0.09075277671217918, "clip_ratio/region_mean": 0.14145267941057682, "reward_total_mean": 0.5564401149749756, "reward_meter_mean": 0.9921140670776367, "reward_meter_std": 0.005870350170880556, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8212906718254089, "reward_repeat_soft_std": 0.18632890284061432, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.1524970829486847, "reward_total_composite_mean": 0.5564401149749756, "reward_total_composite_std": 0.10963989049196243} {"timestamp_utc": "2026-04-13T08:27:54Z", "mode": "train", "global_step": 446, "epoch": 0.04480160723254646, "loss": -0.0893, "grad_norm": 5.285683631896973, "learning_rate": 8.651515151515152e-06, "num_tokens": 808322.0, "completions/mean_length": 97.25, "completions/min_length": 34.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 38.0, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.3324628472328186, "rewards/meter/std": 0.3397580683231354, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.995673656463623, "rewards/repeat_soft/std": 0.004901529755443335, "rewards/judge_quality/mean": 0.5974999666213989, "rewards/judge_quality/std": 0.276082307100296, "rewards/total_composite/mean": 0.45673373341560364, "rewards/total_composite/std": 0.24027307331562042, "reward": 0.45673373341560364, "reward_std": 0.24027305841445923, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1777818351984024, "sampling/sampling_logp_difference/max": 1.8305916786193848, "sampling/importance_sampling_ratio/min": 0.16031868755817413, "sampling/importance_sampling_ratio/mean": 1.0151317119598389, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9830478876829147, "clip_ratio/low_mean": 0.07306061871349812, "clip_ratio/low_min": 0.07306061871349812, "clip_ratio/high_mean": 0.08636224828660488, "clip_ratio/high_max": 0.08636224828660488, "clip_ratio/region_mean": 0.159422867000103, "reward_total_mean": 0.45673373341560364, "reward_meter_mean": 0.3324628472328186, "reward_meter_std": 0.3397580683231354, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.995673656463623, "reward_repeat_soft_std": 0.004901529755443335, "reward_judge_quality_mean": 0.5974999666213989, "reward_judge_quality_std": 0.276082307100296, "reward_total_composite_mean": 0.45673373341560364, "reward_total_composite_std": 0.24027307331562042} {"timestamp_utc": "2026-04-13T08:28:01Z", "mode": "train", "global_step": 447, "epoch": 0.04490205926670015, "loss": 0.0263, "grad_norm": 16.588058471679688, "learning_rate": 8.64848484848485e-06, "num_tokens": 810082.0, "completions/mean_length": 37.0, "completions/min_length": 33.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.0, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.6714106798171997, "rewards/meter/std": 0.3686300814151764, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9947012066841125, "rewards/repeat_soft/std": 0.005062811076641083, "rewards/judge_quality/mean": 0.41874998807907104, "rewards/judge_quality/std": 0.21931305527687073, "rewards/total_composite/mean": 0.5521893501281738, "rewards/total_composite/std": 0.18018938601016998, "reward": 0.5521893501281738, "reward_std": 0.1801893711090088, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23548410832881927, "sampling/sampling_logp_difference/max": 1.2349090576171875, "sampling/importance_sampling_ratio/min": 0.29086118936538696, "sampling/importance_sampling_ratio/mean": 1.0399689674377441, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.829873412847519, "clip_ratio/low_mean": 0.1163807213306427, "clip_ratio/low_min": 0.1163807213306427, "clip_ratio/high_mean": 0.12436641566455364, "clip_ratio/high_max": 0.12436641566455364, "clip_ratio/region_mean": 0.24074713699519634, "reward_total_mean": 0.5521893501281738, "reward_meter_mean": 0.6714106798171997, "reward_meter_std": 0.3686300814151764, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9947012066841125, "reward_repeat_soft_std": 0.005062811076641083, "reward_judge_quality_mean": 0.41874998807907104, "reward_judge_quality_std": 0.21931305527687073, "reward_total_composite_mean": 0.5521893501281738, "reward_total_composite_std": 0.18018938601016998} {"timestamp_utc": "2026-04-13T08:28:08Z", "mode": "train", "global_step": 448, "epoch": 0.045002511300853844, "loss": 0.0682, "grad_norm": 18.4505615234375, "learning_rate": 8.645454545454545e-06, "num_tokens": 811625.0, "completions/mean_length": 33.875, "completions/min_length": 15.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.875, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.8039917945861816, "rewards/meter/std": 0.26347020268440247, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9875937700271606, "rewards/repeat_soft/std": 0.015741875395178795, "rewards/judge_quality/mean": 0.48625001311302185, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.5737026929855347, "rewards/total_composite/std": 0.10885868966579437, "reward": 0.5737026929855347, "reward_std": 0.10885868221521378, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22540470957756042, "sampling/sampling_logp_difference/max": 1.8954334259033203, "sampling/importance_sampling_ratio/min": 0.15025319159030914, "sampling/importance_sampling_ratio/mean": 1.0656899213790894, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4711096584796906, "clip_ratio/low_mean": 0.07114621624350548, "clip_ratio/low_min": 0.07114621624350548, "clip_ratio/high_mean": 0.08506806381046772, "clip_ratio/high_max": 0.08506806381046772, "clip_ratio/region_mean": 0.1562142800539732, "reward_total_mean": 0.5737026929855347, "reward_meter_mean": 0.8039917945861816, "reward_meter_std": 0.26347020268440247, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9875937700271606, "reward_repeat_soft_std": 0.015741875395178795, "reward_judge_quality_mean": 0.48625001311302185, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.5737026929855347, "reward_total_composite_std": 0.10885868966579437} {"timestamp_utc": "2026-04-13T08:28:15Z", "mode": "train", "global_step": 449, "epoch": 0.04510296333500753, "loss": 0.0277, "grad_norm": 11.545234680175781, "learning_rate": 8.642424242424242e-06, "num_tokens": 813078.0, "completions/mean_length": 28.625, "completions/min_length": 23.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.625, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.6695095896720886, "rewards/meter/std": 0.3916354775428772, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9931644797325134, "rewards/repeat_soft/std": 0.012399251572787762, "rewards/judge_quality/mean": 0.5762499570846558, "rewards/judge_quality/std": 0.21413865685462952, "rewards/total_composite/mean": 0.587734580039978, "rewards/total_composite/std": 0.20412905514240265, "reward": 0.587734580039978, "reward_std": 0.20412907004356384, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10341665893793106, "sampling/sampling_logp_difference/max": 1.1627860069274902, "sampling/importance_sampling_ratio/min": 0.3126140236854553, "sampling/importance_sampling_ratio/mean": 1.013893961906433, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7491747736930847, "clip_ratio/low_mean": 0.010416666977107525, "clip_ratio/low_min": 0.010416666977107525, "clip_ratio/high_mean": 0.06982870260253549, "clip_ratio/high_max": 0.06982870260253549, "clip_ratio/region_mean": 0.08024536957964301, "reward_total_mean": 0.587734580039978, "reward_meter_mean": 0.6695095896720886, "reward_meter_std": 0.3916354775428772, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9931644797325134, "reward_repeat_soft_std": 0.012399251572787762, "reward_judge_quality_mean": 0.5762499570846558, "reward_judge_quality_std": 0.21413865685462952, "reward_total_composite_mean": 0.587734580039978, "reward_total_composite_std": 0.20412905514240265} {"timestamp_utc": "2026-04-13T08:28:23Z", "mode": "train", "global_step": 450, "epoch": 0.045203415369161226, "loss": 0.0364, "grad_norm": 23.196592330932617, "learning_rate": 8.63939393939394e-06, "num_tokens": 814546.0, "completions/mean_length": 33.5, "completions/min_length": 28.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.5, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.8268553018569946, "rewards/meter/std": 0.3144039213657379, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9936370253562927, "rewards/repeat_soft/std": 0.006234763655811548, "rewards/judge_quality/mean": 0.4724999964237213, "rewards/judge_quality/std": 0.11310552060604095, "rewards/total_composite/mean": 0.6005456447601318, "rewards/total_composite/std": 0.09474692493677139, "reward": 0.6005456447601318, "reward_std": 0.09474693238735199, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2116003930568695, "sampling/sampling_logp_difference/max": 1.4157276153564453, "sampling/importance_sampling_ratio/min": 0.2427489310503006, "sampling/importance_sampling_ratio/mean": 1.039547324180603, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1985170394182205, "clip_ratio/low_mean": 0.01953125, "clip_ratio/low_min": 0.01953125, "clip_ratio/high_mean": 0.1654675779864192, "clip_ratio/high_max": 0.1654675779864192, "clip_ratio/region_mean": 0.1849988279864192, "reward_total_mean": 0.6005456447601318, "reward_meter_mean": 0.8268553018569946, "reward_meter_std": 0.3144039213657379, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9936370253562927, "reward_repeat_soft_std": 0.006234763655811548, "reward_judge_quality_mean": 0.4724999964237213, "reward_judge_quality_std": 0.11310552060604095, "reward_total_composite_mean": 0.6005456447601318, "reward_total_composite_std": 0.09474692493677139} {"timestamp_utc": "2026-04-13T08:29:42Z", "mode": "eval", "global_step": 450, "epoch": 0.045203415369161226, "eval_loss": NaN, "eval_runtime": 79.4198, "eval_samples_per_second": 1.007, "eval_steps_per_second": 0.126, "eval_num_tokens": 814546.0, "eval_completions/mean_length": 103.1, "eval_completions/min_length": 26.1, "eval_completions/max_length": 361.3, "eval_completions/clipped_ratio": 0.0875, "eval_completions/mean_terminated_length": 63.70952453613281, "eval_completions/min_terminated_length": 26.1, "eval_completions/max_terminated_length": 134.8, "eval_rewards/meter/mean": 0.5790894836187362, "eval_rewards/meter/std": 0.3777293890714645, "eval_rewards/count_adherence/mean": 0.9304166555404663, "eval_rewards/count_adherence/std": 0.12106326185166835, "eval_rewards/hard_gate/mean": 0.875, "eval_rewards/hard_gate/std": 0.28029437065124513, "eval_rewards/repeat_soft/mean": 0.9575423061847687, "eval_rewards/repeat_soft/std": 0.08738601845689117, "eval_rewards/judge_quality/mean": 0.4236250013113022, "eval_rewards/judge_quality/std": 0.19241945892572404, "eval_rewards/total_composite/mean": 0.4662149131298065, "eval_rewards/total_composite/std": 0.22698013633489608, "eval_reward": 0.4662149131298065, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.15265945792198182, "eval_sampling/sampling_logp_difference/max": 1.12666597366333, "eval_sampling/importance_sampling_ratio/min": 0.32656060755252836, "eval_sampling/importance_sampling_ratio/mean": 1.0445361137390137, "eval_sampling/importance_sampling_ratio/max": 1.7058947801589965, "eval_entropy": 2.8295164823532106, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.4662149131298065, "eval_reward_meter_mean": 0.5790894836187362, "eval_reward_meter_std": 0.3777293890714645, "eval_reward_count_adherence_mean": 0.9304166555404663, "eval_reward_count_adherence_std": 0.12106326185166835, "eval_reward_hard_gate_mean": 0.875, "eval_reward_hard_gate_std": 0.28029437065124513, "eval_reward_repeat_soft_mean": 0.9575423061847687, "eval_reward_repeat_soft_std": 0.08738601845689117, "eval_reward_judge_quality_mean": 0.4236250013113022, "eval_reward_judge_quality_std": 0.19241945892572404, "eval_reward_total_composite_mean": 0.4662149131298065, "eval_reward_total_composite_std": 0.22698013633489608} {"timestamp_utc": "2026-04-13T08:29:53Z", "mode": "train", "global_step": 451, "epoch": 0.045303867403314914, "loss": 0.0657, "grad_norm": 14.971288681030273, "learning_rate": 8.636363636363637e-06, "num_tokens": 816493.0, "completions/mean_length": 61.375, "completions/min_length": 57.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.375, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.7840032577514648, "rewards/meter/std": 0.33735308051109314, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9903982281684875, "rewards/repeat_soft/std": 0.009481220506131649, "rewards/judge_quality/mean": 0.5325000286102295, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.6165653467178345, "rewards/total_composite/std": 0.1414032131433487, "reward": 0.6165653467178345, "reward_std": 0.1414031982421875, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23850522935390472, "sampling/sampling_logp_difference/max": 1.3375978469848633, "sampling/importance_sampling_ratio/min": 0.2624754309654236, "sampling/importance_sampling_ratio/mean": 1.0306938886642456, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.7483876645565033, "clip_ratio/low_mean": 0.1360711269080639, "clip_ratio/low_min": 0.1360711269080639, "clip_ratio/high_mean": 0.08889990113675594, "clip_ratio/high_max": 0.08889990113675594, "clip_ratio/region_mean": 0.22497102804481983, "reward_total_mean": 0.6165653467178345, "reward_meter_mean": 0.7840032577514648, "reward_meter_std": 0.33735308051109314, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9903982281684875, "reward_repeat_soft_std": 0.009481220506131649, "reward_judge_quality_mean": 0.5325000286102295, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.6165653467178345, "reward_total_composite_std": 0.1414032131433487} {"timestamp_utc": "2026-04-13T08:30:05Z", "mode": "train", "global_step": 452, "epoch": 0.04540431943746861, "loss": -0.1426, "grad_norm": 4.067929267883301, "learning_rate": 8.633333333333334e-06, "num_tokens": 818955.0, "completions/mean_length": 154.75, "completions/min_length": 97.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 103.71428680419922, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.45677509903907776, "rewards/meter/std": 0.34570378065109253, "rewards/count_adherence/mean": 0.9583333134651184, "rewards/count_adherence/std": 0.07715168595314026, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.946021556854248, "rewards/repeat_soft/std": 0.0689612403512001, "rewards/judge_quality/mean": 0.4112499952316284, "rewards/judge_quality/std": 0.1797965168952942, "rewards/total_composite/mean": 0.4367862939834595, "rewards/total_composite/std": 0.22509154677391052, "reward": 0.4367862939834595, "reward_std": 0.22509156167507172, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19598762691020966, "sampling/sampling_logp_difference/max": 1.459304928779602, "sampling/importance_sampling_ratio/min": 0.23239775002002716, "sampling/importance_sampling_ratio/mean": 1.031680941581726, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5597435683012009, "clip_ratio/low_mean": 0.05722092464566231, "clip_ratio/low_min": 0.05722092464566231, "clip_ratio/high_mean": 0.0950970258563757, "clip_ratio/high_max": 0.0950970258563757, "clip_ratio/region_mean": 0.152317950502038, "reward_total_mean": 0.4367862939834595, "reward_meter_mean": 0.45677509903907776, "reward_meter_std": 0.34570378065109253, "reward_count_adherence_mean": 0.9583333134651184, "reward_count_adherence_std": 0.07715168595314026, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.946021556854248, "reward_repeat_soft_std": 0.0689612403512001, "reward_judge_quality_mean": 0.4112499952316284, "reward_judge_quality_std": 0.1797965168952942, "reward_total_composite_mean": 0.4367862939834595, "reward_total_composite_std": 0.22509154677391052} {"timestamp_utc": "2026-04-13T08:30:17Z", "mode": "train", "global_step": 453, "epoch": 0.0455047714716223, "loss": -0.0817, "grad_norm": 3.1986804008483887, "learning_rate": 8.630303030303032e-06, "num_tokens": 820538.0, "completions/mean_length": 160.875, "completions/min_length": 37.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 43.833335876464844, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.7236465811729431, "rewards/meter/std": 0.4409694969654083, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.9389361143112183, "rewards/repeat_soft/std": 0.0779365822672844, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.2564002275466919, "rewards/total_composite/mean": 0.4048718810081482, "rewards/total_composite/std": 0.3453499972820282, "reward": 0.4048718810081482, "reward_std": 0.3453499972820282, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1954086422920227, "sampling/sampling_logp_difference/max": 1.4350799322128296, "sampling/importance_sampling_ratio/min": 0.23809632658958435, "sampling/importance_sampling_ratio/mean": 1.0326827764511108, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5149724185466766, "clip_ratio/low_mean": 0.027027027681469917, "clip_ratio/low_min": 0.027027027681469917, "clip_ratio/high_mean": 0.1101987762376666, "clip_ratio/high_max": 0.1101987762376666, "clip_ratio/region_mean": 0.13722580391913652, "reward_total_mean": 0.4048718810081482, "reward_meter_mean": 0.7236465811729431, "reward_meter_std": 0.4409694969654083, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.9389361143112183, "reward_repeat_soft_std": 0.0779365822672844, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.2564002275466919, "reward_total_composite_mean": 0.4048718810081482, "reward_total_composite_std": 0.3453499972820282} {"timestamp_utc": "2026-04-13T08:30:25Z", "mode": "train", "global_step": 454, "epoch": 0.04560522350577599, "loss": 0.0339, "grad_norm": 15.092330932617188, "learning_rate": 8.627272727272727e-06, "num_tokens": 822099.0, "completions/mean_length": 31.125, "completions/min_length": 29.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.125, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9551911950111389, "rewards/meter/std": 0.04419620335102081, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9948406219482422, "rewards/repeat_soft/std": 0.004915553145110607, "rewards/judge_quality/mean": 0.5475000143051147, "rewards/judge_quality/std": 0.21822334825992584, "rewards/total_composite/mean": 0.690136194229126, "rewards/total_composite/std": 0.14133980870246887, "reward": 0.690136194229126, "reward_std": 0.14133977890014648, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18606992065906525, "sampling/sampling_logp_difference/max": 1.713277816772461, "sampling/importance_sampling_ratio/min": 0.18027392029762268, "sampling/importance_sampling_ratio/mean": 1.0348671674728394, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6780478358268738, "clip_ratio/low_mean": 0.12285635527223349, "clip_ratio/low_min": 0.12285635527223349, "clip_ratio/high_mean": 0.04359878972172737, "clip_ratio/high_max": 0.04359878972172737, "clip_ratio/region_mean": 0.16645514499396086, "reward_total_mean": 0.690136194229126, "reward_meter_mean": 0.9551911950111389, "reward_meter_std": 0.04419620335102081, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9948406219482422, "reward_repeat_soft_std": 0.004915553145110607, "reward_judge_quality_mean": 0.5475000143051147, "reward_judge_quality_std": 0.21822334825992584, "reward_total_composite_mean": 0.690136194229126, "reward_total_composite_std": 0.14133980870246887} {"timestamp_utc": "2026-04-13T08:30:38Z", "mode": "train", "global_step": 455, "epoch": 0.045705675539929685, "loss": -0.1641, "grad_norm": 2.153820514678955, "learning_rate": 8.624242424242424e-06, "num_tokens": 824073.0, "completions/mean_length": 126.75, "completions/min_length": 60.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 71.71428680419922, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.8764752745628357, "rewards/meter/std": 0.22269082069396973, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9060655832290649, "rewards/repeat_soft/std": 0.10061728209257126, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.13593590259552002, "rewards/total_composite/mean": 0.5059380531311035, "rewards/total_composite/std": 0.20810282230377197, "reward": 0.5059380531311035, "reward_std": 0.20810280740261078, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20393435657024384, "sampling/sampling_logp_difference/max": 1.2285990715026855, "sampling/importance_sampling_ratio/min": 0.2927023470401764, "sampling/importance_sampling_ratio/mean": 1.0643165111541748, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.230049356818199, "clip_ratio/low_mean": 0.017473118379712105, "clip_ratio/low_min": 0.017473118379712105, "clip_ratio/high_mean": 0.15428759716451168, "clip_ratio/high_max": 0.15428759716451168, "clip_ratio/region_mean": 0.17176071554422379, "reward_total_mean": 0.5059380531311035, "reward_meter_mean": 0.8764752745628357, "reward_meter_std": 0.22269082069396973, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9060655832290649, "reward_repeat_soft_std": 0.10061728209257126, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.13593590259552002, "reward_total_composite_mean": 0.5059380531311035, "reward_total_composite_std": 0.20810282230377197} {"timestamp_utc": "2026-04-13T08:30:46Z", "mode": "train", "global_step": 456, "epoch": 0.04580612757408337, "loss": -0.0041, "grad_norm": 15.241849899291992, "learning_rate": 8.621212121212122e-06, "num_tokens": 825872.0, "completions/mean_length": 50.875, "completions/min_length": 45.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.875, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.5705286264419556, "rewards/meter/std": 0.4435427486896515, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9980649948120117, "rewards/repeat_soft/std": 0.0014921160181984305, "rewards/judge_quality/mean": 0.8575000166893005, "rewards/judge_quality/std": 0.176776722073555, "rewards/total_composite/mean": 0.6546416282653809, "rewards/total_composite/std": 0.25483405590057373, "reward": 0.6546416282653809, "reward_std": 0.25483405590057373, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10343720763921738, "sampling/sampling_logp_difference/max": 1.5640950202941895, "sampling/importance_sampling_ratio/min": 0.20927731692790985, "sampling/importance_sampling_ratio/mean": 0.9993636608123779, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4827927704900503, "clip_ratio/low_mean": 0.06564394384622574, "clip_ratio/low_min": 0.06564394384622574, "clip_ratio/high_mean": 0.03059012140147388, "clip_ratio/high_max": 0.03059012140147388, "clip_ratio/region_mean": 0.09623406524769962, "reward_total_mean": 0.6546416282653809, "reward_meter_mean": 0.5705286264419556, "reward_meter_std": 0.4435427486896515, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9980649948120117, "reward_repeat_soft_std": 0.0014921160181984305, "reward_judge_quality_mean": 0.8575000166893005, "reward_judge_quality_std": 0.176776722073555, "reward_total_composite_mean": 0.6546416282653809, "reward_total_composite_std": 0.25483405590057373} {"timestamp_utc": "2026-04-13T08:30:58Z", "mode": "train", "global_step": 457, "epoch": 0.04590657960823707, "loss": -0.0919, "grad_norm": 3.6754684448242188, "learning_rate": 8.618181818181819e-06, "num_tokens": 827262.0, "completions/mean_length": 93.75, "completions/min_length": 27.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 34.0, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.5208476185798645, "rewards/meter/std": 0.33431369066238403, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.991050124168396, "rewards/repeat_soft/std": 0.012516842223703861, "rewards/judge_quality/mean": 0.45249998569488525, "rewards/judge_quality/std": 0.2665520906448364, "rewards/total_composite/mean": 0.44626879692077637, "rewards/total_composite/std": 0.20548522472381592, "reward": 0.44626879692077637, "reward_std": 0.20548520982265472, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2468198984861374, "sampling/sampling_logp_difference/max": 1.9478073120117188, "sampling/importance_sampling_ratio/min": 0.1425863802433014, "sampling/importance_sampling_ratio/mean": 1.025658130645752, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4353190809488297, "clip_ratio/low_mean": 0.04750503972172737, "clip_ratio/low_min": 0.04750503972172737, "clip_ratio/high_mean": 0.12348518334329128, "clip_ratio/high_max": 0.12348518334329128, "clip_ratio/region_mean": 0.17099022306501865, "reward_total_mean": 0.44626879692077637, "reward_meter_mean": 0.5208476185798645, "reward_meter_std": 0.33431369066238403, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.991050124168396, "reward_repeat_soft_std": 0.012516842223703861, "reward_judge_quality_mean": 0.45249998569488525, "reward_judge_quality_std": 0.2665520906448364, "reward_total_composite_mean": 0.44626879692077637, "reward_total_composite_std": 0.20548522472381592} {"timestamp_utc": "2026-04-13T08:31:04Z", "mode": "train", "global_step": 458, "epoch": 0.046007031642390755, "loss": -0.0244, "grad_norm": 16.54827308654785, "learning_rate": 8.615151515151516e-06, "num_tokens": 828726.0, "completions/mean_length": 26.0, "completions/min_length": 18.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.0, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.6205868124961853, "rewards/meter/std": 0.37930798530578613, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9410344362258911, "rewards/repeat_soft/std": 0.03412774205207825, "rewards/judge_quality/mean": 0.5487500429153442, "rewards/judge_quality/std": 0.22937415540218353, "rewards/total_composite/mean": 0.5913739204406738, "rewards/total_composite/std": 0.22341004014015198, "reward": 0.5913739204406738, "reward_std": 0.22341004014015198, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1890912503004074, "sampling/sampling_logp_difference/max": 0.9961767196655273, "sampling/importance_sampling_ratio/min": 0.3692886531352997, "sampling/importance_sampling_ratio/mean": 1.0250221490859985, "sampling/importance_sampling_ratio/max": 1.89702570438385, "entropy": 1.8906628489494324, "clip_ratio/low_mean": 0.09209030866622925, "clip_ratio/low_min": 0.09209030866622925, "clip_ratio/high_mean": 0.06419101729989052, "clip_ratio/high_max": 0.06419101729989052, "clip_ratio/region_mean": 0.15628132596611977, "reward_total_mean": 0.5913739204406738, "reward_meter_mean": 0.6205868124961853, "reward_meter_std": 0.37930798530578613, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9410344362258911, "reward_repeat_soft_std": 0.03412774205207825, "reward_judge_quality_mean": 0.5487500429153442, "reward_judge_quality_std": 0.22937415540218353, "reward_total_composite_mean": 0.5913739204406738, "reward_total_composite_std": 0.22341004014015198} {"timestamp_utc": "2026-04-13T08:31:12Z", "mode": "train", "global_step": 459, "epoch": 0.04610748367654445, "loss": 0.1777, "grad_norm": 17.388628005981445, "learning_rate": 8.612121212121213e-06, "num_tokens": 830349.0, "completions/mean_length": 39.875, "completions/min_length": 30.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.875, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.9704955816268921, "rewards/meter/std": 0.03484657034277916, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9952265024185181, "rewards/repeat_soft/std": 0.007804577238857746, "rewards/judge_quality/mean": 0.5687500238418579, "rewards/judge_quality/std": 0.19111983478069305, "rewards/total_composite/mean": 0.7075623273849487, "rewards/total_composite/std": 0.12064624577760696, "reward": 0.7075623273849487, "reward_std": 0.12064624577760696, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21925927698612213, "sampling/sampling_logp_difference/max": 1.3226232528686523, "sampling/importance_sampling_ratio/min": 0.2664354741573334, "sampling/importance_sampling_ratio/mean": 1.0799548625946045, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6135420203208923, "clip_ratio/low_mean": 0.11544453538954258, "clip_ratio/low_min": 0.11544453538954258, "clip_ratio/high_mean": 0.06963716261088848, "clip_ratio/high_max": 0.06963716261088848, "clip_ratio/region_mean": 0.18508169800043106, "reward_total_mean": 0.7075623273849487, "reward_meter_mean": 0.9704955816268921, "reward_meter_std": 0.03484657034277916, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9952265024185181, "reward_repeat_soft_std": 0.007804577238857746, "reward_judge_quality_mean": 0.5687500238418579, "reward_judge_quality_std": 0.19111983478069305, "reward_total_composite_mean": 0.7075623273849487, "reward_total_composite_std": 0.12064624577760696} {"timestamp_utc": "2026-04-13T08:31:21Z", "mode": "train", "global_step": 460, "epoch": 0.046207935710698145, "loss": -0.0999, "grad_norm": 13.907397270202637, "learning_rate": 8.60909090909091e-06, "num_tokens": 832391.0, "completions/mean_length": 71.25, "completions/min_length": 37.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.25, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.5582280158996582, "rewards/meter/std": 0.28584811091423035, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9957082271575928, "rewards/repeat_soft/std": 0.004323306959122419, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.49182212352752686, "rewards/total_composite/std": 0.09495563805103302, "reward": 0.49182212352752686, "reward_std": 0.09495565295219421, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1721809357404709, "sampling/sampling_logp_difference/max": 1.770277976989746, "sampling/importance_sampling_ratio/min": 0.1702856421470642, "sampling/importance_sampling_ratio/mean": 1.0064765214920044, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.007186632603407, "clip_ratio/low_mean": 0.06141394283622503, "clip_ratio/low_min": 0.06141394283622503, "clip_ratio/high_mean": 0.10252820141613483, "clip_ratio/high_max": 0.10252820141613483, "clip_ratio/region_mean": 0.16394214425235987, "reward_total_mean": 0.49182212352752686, "reward_meter_mean": 0.5582280158996582, "reward_meter_std": 0.28584811091423035, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9957082271575928, "reward_repeat_soft_std": 0.004323306959122419, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.49182212352752686, "reward_total_composite_std": 0.09495563805103302} {"timestamp_utc": "2026-04-13T08:31:28Z", "mode": "train", "global_step": 461, "epoch": 0.04630838774485183, "loss": -0.057, "grad_norm": 25.16307258605957, "learning_rate": 8.606060606060606e-06, "num_tokens": 833928.0, "completions/mean_length": 28.125, "completions/min_length": 21.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.125, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.7145506739616394, "rewards/meter/std": 0.358488529920578, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.995843231678009, "rewards/repeat_soft/std": 0.0051807500422000885, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.5824846029281616, "rewards/total_composite/std": 0.16257049143314362, "reward": 0.5824846029281616, "reward_std": 0.16257049143314362, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19297046959400177, "sampling/sampling_logp_difference/max": 1.3607921600341797, "sampling/importance_sampling_ratio/min": 0.3609359860420227, "sampling/importance_sampling_ratio/mean": 1.037373661994934, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3269875720143318, "clip_ratio/low_mean": 0.13090106844902039, "clip_ratio/low_min": 0.13090106844902039, "clip_ratio/high_mean": 0.0990843316540122, "clip_ratio/high_max": 0.0990843316540122, "clip_ratio/region_mean": 0.2299854001030326, "reward_total_mean": 0.5824846029281616, "reward_meter_mean": 0.7145506739616394, "reward_meter_std": 0.358488529920578, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.995843231678009, "reward_repeat_soft_std": 0.0051807500422000885, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.5824846029281616, "reward_total_composite_std": 0.16257049143314362} {"timestamp_utc": "2026-04-13T08:31:36Z", "mode": "train", "global_step": 462, "epoch": 0.04640883977900553, "loss": 0.0167, "grad_norm": 11.881519317626953, "learning_rate": 8.603030303030303e-06, "num_tokens": 835639.0, "completions/mean_length": 50.875, "completions/min_length": 47.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.875, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.8157943487167358, "rewards/meter/std": 0.2549947202205658, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9945453405380249, "rewards/repeat_soft/std": 0.005802301224321127, "rewards/judge_quality/mean": 0.5325000286102295, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.6300034523010254, "rewards/total_composite/std": 0.11230958998203278, "reward": 0.6300034523010254, "reward_std": 0.11230959743261337, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17807799577713013, "sampling/sampling_logp_difference/max": 1.3732337951660156, "sampling/importance_sampling_ratio/min": 0.25328654050827026, "sampling/importance_sampling_ratio/mean": 1.0447531938552856, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8934835344552994, "clip_ratio/low_mean": 0.10798620246350765, "clip_ratio/low_min": 0.10798620246350765, "clip_ratio/high_mean": 0.02238159440457821, "clip_ratio/high_max": 0.02238159440457821, "clip_ratio/region_mean": 0.13036779686808586, "reward_total_mean": 0.6300034523010254, "reward_meter_mean": 0.8157943487167358, "reward_meter_std": 0.2549947202205658, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9945453405380249, "reward_repeat_soft_std": 0.005802301224321127, "reward_judge_quality_mean": 0.5325000286102295, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.6300034523010254, "reward_total_composite_std": 0.11230958998203278} {"timestamp_utc": "2026-04-13T08:31:43Z", "mode": "train", "global_step": 463, "epoch": 0.046509291813159215, "loss": -0.0119, "grad_norm": 16.91208839416504, "learning_rate": 8.6e-06, "num_tokens": 837146.0, "completions/mean_length": 35.375, "completions/min_length": 31.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.375, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.6957108974456787, "rewards/meter/std": 0.32157692313194275, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9874091148376465, "rewards/repeat_soft/std": 0.015234079211950302, "rewards/judge_quality/mean": 0.7987500429153442, "rewards/judge_quality/std": 0.22465451061725616, "rewards/total_composite/mean": 0.7096598148345947, "rewards/total_composite/std": 0.20204918086528778, "reward": 0.7096598148345947, "reward_std": 0.2020491659641266, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20317187905311584, "sampling/sampling_logp_difference/max": 1.7107229232788086, "sampling/importance_sampling_ratio/min": 0.18073508143424988, "sampling/importance_sampling_ratio/mean": 1.0086177587509155, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5891913622617722, "clip_ratio/low_mean": 0.03721730224788189, "clip_ratio/low_min": 0.03721730224788189, "clip_ratio/high_mean": 0.10852282680571079, "clip_ratio/high_max": 0.10852282680571079, "clip_ratio/region_mean": 0.14574012905359268, "reward_total_mean": 0.7096598148345947, "reward_meter_mean": 0.6957108974456787, "reward_meter_std": 0.32157692313194275, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9874091148376465, "reward_repeat_soft_std": 0.015234079211950302, "reward_judge_quality_mean": 0.7987500429153442, "reward_judge_quality_std": 0.22465451061725616, "reward_total_composite_mean": 0.7096598148345947, "reward_total_composite_std": 0.20204918086528778} {"timestamp_utc": "2026-04-13T08:31:50Z", "mode": "train", "global_step": 464, "epoch": 0.04660974384731291, "loss": 0.054, "grad_norm": 12.949996948242188, "learning_rate": 8.596969696969698e-06, "num_tokens": 838720.0, "completions/mean_length": 32.75, "completions/min_length": 31.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.75, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.5962459444999695, "rewards/meter/std": 0.33634284138679504, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.964966893196106, "rewards/repeat_soft/std": 0.011403634212911129, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.5142271518707275, "rewards/total_composite/std": 0.08487378060817719, "reward": 0.5142271518707275, "reward_std": 0.08487378060817719, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09167084097862244, "sampling/sampling_logp_difference/max": 1.1023716926574707, "sampling/importance_sampling_ratio/min": 0.3320825397968292, "sampling/importance_sampling_ratio/mean": 1.03141450881958, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7570228986442089, "clip_ratio/low_mean": 0.030455619795247912, "clip_ratio/low_min": 0.030455619795247912, "clip_ratio/high_mean": 0.05427376227453351, "clip_ratio/high_max": 0.05427376227453351, "clip_ratio/region_mean": 0.08472938206978142, "reward_total_mean": 0.5142271518707275, "reward_meter_mean": 0.5962459444999695, "reward_meter_std": 0.33634284138679504, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.964966893196106, "reward_repeat_soft_std": 0.011403634212911129, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.5142271518707275, "reward_total_composite_std": 0.08487378060817719} {"timestamp_utc": "2026-04-13T08:31:57Z", "mode": "train", "global_step": 465, "epoch": 0.0467101958814666, "loss": 0.0338, "grad_norm": 19.132905960083008, "learning_rate": 8.593939393939395e-06, "num_tokens": 839982.0, "completions/mean_length": 20.75, "completions/min_length": 17.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 20.75, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.7611384391784668, "rewards/meter/std": 0.33914369344711304, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9602519273757935, "rewards/repeat_soft/std": 0.006358357612043619, "rewards/judge_quality/mean": 0.36374998092651367, "rewards/judge_quality/std": 0.09500939399003983, "rewards/total_composite/mean": 0.5303007960319519, "rewards/total_composite/std": 0.09380699694156647, "reward": 0.5303007960319519, "reward_std": 0.09380699694156647, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20365390181541443, "sampling/sampling_logp_difference/max": 0.8794517517089844, "sampling/importance_sampling_ratio/min": 0.41501039266586304, "sampling/importance_sampling_ratio/mean": 1.028915524482727, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1659615635871887, "clip_ratio/low_mean": 0.11045639030635357, "clip_ratio/low_min": 0.11045639030635357, "clip_ratio/high_mean": 0.1121031790971756, "clip_ratio/high_max": 0.1121031790971756, "clip_ratio/region_mean": 0.22255956940352917, "reward_total_mean": 0.5303007960319519, "reward_meter_mean": 0.7611384391784668, "reward_meter_std": 0.33914369344711304, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9602519273757935, "reward_repeat_soft_std": 0.006358357612043619, "reward_judge_quality_mean": 0.36374998092651367, "reward_judge_quality_std": 0.09500939399003983, "reward_total_composite_mean": 0.5303007960319519, "reward_total_composite_std": 0.09380699694156647} {"timestamp_utc": "2026-04-13T08:32:05Z", "mode": "train", "global_step": 466, "epoch": 0.04681064791562029, "loss": -0.0183, "grad_norm": 14.528681755065918, "learning_rate": 8.590909090909092e-06, "num_tokens": 841861.0, "completions/mean_length": 69.875, "completions/min_length": 59.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.875, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.7092288136482239, "rewards/meter/std": 0.3216937780380249, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.981717586517334, "rewards/repeat_soft/std": 0.01015958096832037, "rewards/judge_quality/mean": 0.6575000286102295, "rewards/judge_quality/std": 0.2133910059928894, "rewards/total_composite/mean": 0.6228625178337097, "rewards/total_composite/std": 0.11276643723249435, "reward": 0.6228625178337097, "reward_std": 0.11276643723249435, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19361798465251923, "sampling/sampling_logp_difference/max": 1.257155418395996, "sampling/importance_sampling_ratio/min": 0.284462034702301, "sampling/importance_sampling_ratio/mean": 1.029369831085205, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9872581958770752, "clip_ratio/low_mean": 0.1122061163187027, "clip_ratio/low_min": 0.1122061163187027, "clip_ratio/high_mean": 0.042305586859583855, "clip_ratio/high_max": 0.042305586859583855, "clip_ratio/region_mean": 0.15451170317828655, "reward_total_mean": 0.6228625178337097, "reward_meter_mean": 0.7092288136482239, "reward_meter_std": 0.3216937780380249, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.981717586517334, "reward_repeat_soft_std": 0.01015958096832037, "reward_judge_quality_mean": 0.6575000286102295, "reward_judge_quality_std": 0.2133910059928894, "reward_total_composite_mean": 0.6228625178337097, "reward_total_composite_std": 0.11276643723249435} {"timestamp_utc": "2026-04-13T08:32:12Z", "mode": "train", "global_step": 467, "epoch": 0.04691109994977398, "loss": 0.1651, "grad_norm": 28.204408645629883, "learning_rate": 8.587878787878788e-06, "num_tokens": 843422.0, "completions/mean_length": 30.125, "completions/min_length": 23.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.125, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.6777348518371582, "rewards/meter/std": 0.3601137399673462, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9606072902679443, "rewards/repeat_soft/std": 0.037384357303380966, "rewards/judge_quality/mean": 0.5887500047683716, "rewards/judge_quality/std": 0.22183892130851746, "rewards/total_composite/mean": 0.5091491937637329, "rewards/total_composite/std": 0.26229557394981384, "reward": 0.5091491937637329, "reward_std": 0.26229560375213623, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2445119023323059, "sampling/sampling_logp_difference/max": 2.081390380859375, "sampling/importance_sampling_ratio/min": 0.12475662678480148, "sampling/importance_sampling_ratio/mean": 1.021946668624878, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9585850909352303, "clip_ratio/low_mean": 0.08880617469549179, "clip_ratio/low_min": 0.08880617469549179, "clip_ratio/high_mean": 0.15226692333817482, "clip_ratio/high_max": 0.15226692333817482, "clip_ratio/region_mean": 0.2410730980336666, "reward_total_mean": 0.5091491937637329, "reward_meter_mean": 0.6777348518371582, "reward_meter_std": 0.3601137399673462, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9606072902679443, "reward_repeat_soft_std": 0.037384357303380966, "reward_judge_quality_mean": 0.5887500047683716, "reward_judge_quality_std": 0.22183892130851746, "reward_total_composite_mean": 0.5091491937637329, "reward_total_composite_std": 0.26229557394981384} {"timestamp_utc": "2026-04-13T08:32:19Z", "mode": "train", "global_step": 468, "epoch": 0.047011551983927674, "loss": 0.0665, "grad_norm": 21.026941299438477, "learning_rate": 8.584848484848485e-06, "num_tokens": 845212.0, "completions/mean_length": 48.75, "completions/min_length": 46.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.75, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.7666157484054565, "rewards/meter/std": 0.28943267464637756, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9710464477539062, "rewards/repeat_soft/std": 0.027212951332330704, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5549430847167969, "rewards/total_composite/std": 0.07763629406690598, "reward": 0.5549430847167969, "reward_std": 0.07763627916574478, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20255707204341888, "sampling/sampling_logp_difference/max": 1.8685169219970703, "sampling/importance_sampling_ratio/min": 0.15435241162776947, "sampling/importance_sampling_ratio/mean": 0.9967963099479675, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.243774227797985, "clip_ratio/low_mean": 0.04402872361242771, "clip_ratio/low_min": 0.04402872361242771, "clip_ratio/high_mean": 0.1832888089120388, "clip_ratio/high_max": 0.1832888089120388, "clip_ratio/region_mean": 0.22731753252446651, "reward_total_mean": 0.5549430847167969, "reward_meter_mean": 0.7666157484054565, "reward_meter_std": 0.28943267464637756, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9710464477539062, "reward_repeat_soft_std": 0.027212951332330704, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5549430847167969, "reward_total_composite_std": 0.07763629406690598} {"timestamp_utc": "2026-04-13T08:32:27Z", "mode": "train", "global_step": 469, "epoch": 0.04711200401808137, "loss": 0.0522, "grad_norm": 15.375267028808594, "learning_rate": 8.581818181818183e-06, "num_tokens": 846747.0, "completions/mean_length": 35.875, "completions/min_length": 33.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.875, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.594794511795044, "rewards/meter/std": 0.4424631595611572, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9962560534477234, "rewards/repeat_soft/std": 0.00541492085903883, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.13695022463798523, "rewards/total_composite/mean": 0.5491325855255127, "rewards/total_composite/std": 0.1585705280303955, "reward": 0.5491325855255127, "reward_std": 0.1585705429315567, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23154234886169434, "sampling/sampling_logp_difference/max": 1.4478940963745117, "sampling/importance_sampling_ratio/min": 0.2350647896528244, "sampling/importance_sampling_ratio/mean": 1.0410370826721191, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.5242821127176285, "clip_ratio/low_mean": 0.09705086797475815, "clip_ratio/low_min": 0.09705086797475815, "clip_ratio/high_mean": 0.1299495492130518, "clip_ratio/high_max": 0.1299495492130518, "clip_ratio/region_mean": 0.22700041718780994, "reward_total_mean": 0.5491325855255127, "reward_meter_mean": 0.594794511795044, "reward_meter_std": 0.4424631595611572, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9962560534477234, "reward_repeat_soft_std": 0.00541492085903883, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.13695022463798523, "reward_total_composite_mean": 0.5491325855255127, "reward_total_composite_std": 0.1585705280303955} {"timestamp_utc": "2026-04-13T08:32:34Z", "mode": "train", "global_step": 470, "epoch": 0.047212456052235056, "loss": 0.1063, "grad_norm": 18.91813850402832, "learning_rate": 8.57878787878788e-06, "num_tokens": 848342.0, "completions/mean_length": 34.375, "completions/min_length": 31.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.375, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.6956475377082825, "rewards/meter/std": 0.3264574110507965, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9633575081825256, "rewards/repeat_soft/std": 0.0264133308082819, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.5591296553611755, "rewards/total_composite/std": 0.09579839557409286, "reward": 0.5591296553611755, "reward_std": 0.09579840302467346, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19767968356609344, "sampling/sampling_logp_difference/max": 1.4106063842773438, "sampling/importance_sampling_ratio/min": 0.2439952939748764, "sampling/importance_sampling_ratio/mean": 1.0326085090637207, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9085746258497238, "clip_ratio/low_mean": 0.078125, "clip_ratio/low_min": 0.078125, "clip_ratio/high_mean": 0.10592788551002741, "clip_ratio/high_max": 0.10592788551002741, "clip_ratio/region_mean": 0.1840528855100274, "reward_total_mean": 0.5591296553611755, "reward_meter_mean": 0.6956475377082825, "reward_meter_std": 0.3264574110507965, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9633575081825256, "reward_repeat_soft_std": 0.0264133308082819, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.5591296553611755, "reward_total_composite_std": 0.09579839557409286} {"timestamp_utc": "2026-04-13T08:32:41Z", "mode": "train", "global_step": 471, "epoch": 0.04731290808638875, "loss": 0.0715, "grad_norm": 12.654136657714844, "learning_rate": 8.575757575757575e-06, "num_tokens": 850102.0, "completions/mean_length": 57.0, "completions/min_length": 50.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.0, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.5279563665390015, "rewards/meter/std": 0.3748530149459839, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.922204852104187, "rewards/repeat_soft/std": 0.09402043372392654, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.12906257808208466, "rewards/total_composite/mean": 0.44840341806411743, "rewards/total_composite/std": 0.22141452133655548, "reward": 0.44840341806411743, "reward_std": 0.2214145064353943, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21207599341869354, "sampling/sampling_logp_difference/max": 1.8050222396850586, "sampling/importance_sampling_ratio/min": 0.16447080671787262, "sampling/importance_sampling_ratio/mean": 1.0279196500778198, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.282945603132248, "clip_ratio/low_mean": 0.07176537904888391, "clip_ratio/low_min": 0.07176537904888391, "clip_ratio/high_mean": 0.1051339004188776, "clip_ratio/high_max": 0.1051339004188776, "clip_ratio/region_mean": 0.17689927946776152, "reward_total_mean": 0.44840341806411743, "reward_meter_mean": 0.5279563665390015, "reward_meter_std": 0.3748530149459839, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.922204852104187, "reward_repeat_soft_std": 0.09402043372392654, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.12906257808208466, "reward_total_composite_mean": 0.44840341806411743, "reward_total_composite_std": 0.22141452133655548} {"timestamp_utc": "2026-04-13T08:32:48Z", "mode": "train", "global_step": 472, "epoch": 0.04741336012054244, "loss": -0.0355, "grad_norm": 12.875102043151855, "learning_rate": 8.572727272727274e-06, "num_tokens": 851964.0, "completions/mean_length": 57.75, "completions/min_length": 49.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.75, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9655599594116211, "rewards/meter/std": 0.04402674362063408, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8224166631698608, "rewards/repeat_soft/std": 0.2796517312526703, "rewards/judge_quality/mean": 0.4025000035762787, "rewards/judge_quality/std": 0.16446885466575623, "rewards/total_composite/mean": 0.5757567882537842, "rewards/total_composite/std": 0.13585293292999268, "reward": 0.5757567882537842, "reward_std": 0.13585293292999268, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17359167337417603, "sampling/sampling_logp_difference/max": 1.4312171936035156, "sampling/importance_sampling_ratio/min": 0.23901782929897308, "sampling/importance_sampling_ratio/mean": 1.0380702018737793, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.810102954506874, "clip_ratio/low_mean": 0.017087023705244064, "clip_ratio/low_min": 0.017087023705244064, "clip_ratio/high_mean": 0.14845621213316917, "clip_ratio/high_max": 0.14845621213316917, "clip_ratio/region_mean": 0.16554323583841324, "reward_total_mean": 0.5757567882537842, "reward_meter_mean": 0.9655599594116211, "reward_meter_std": 0.04402674362063408, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8224166631698608, "reward_repeat_soft_std": 0.2796517312526703, "reward_judge_quality_mean": 0.4025000035762787, "reward_judge_quality_std": 0.16446885466575623, "reward_total_composite_mean": 0.5757567882537842, "reward_total_composite_std": 0.13585293292999268} {"timestamp_utc": "2026-04-13T08:33:00Z", "mode": "train", "global_step": 473, "epoch": 0.04751381215469613, "loss": -0.0469, "grad_norm": 7.2138872146606445, "learning_rate": 8.56969696969697e-06, "num_tokens": 853783.0, "completions/mean_length": 124.375, "completions/min_length": 63.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 69.0, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.45798543095588684, "rewards/meter/std": 0.24643538892269135, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.9681723713874817, "rewards/repeat_soft/std": 0.033738333731889725, "rewards/judge_quality/mean": 0.39750000834465027, "rewards/judge_quality/std": 0.19031928479671478, "rewards/total_composite/mean": 0.3047120273113251, "rewards/total_composite/std": 0.26664313673973083, "reward": 0.3047120273113251, "reward_std": 0.26664313673973083, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21132715046405792, "sampling/sampling_logp_difference/max": 1.4810504913330078, "sampling/importance_sampling_ratio/min": 0.22739869356155396, "sampling/importance_sampling_ratio/mean": 1.033980131149292, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6572274565696716, "clip_ratio/low_mean": 0.056581297889351845, "clip_ratio/low_min": 0.056581297889351845, "clip_ratio/high_mean": 0.12632135301828384, "clip_ratio/high_max": 0.12632135301828384, "clip_ratio/region_mean": 0.1829026509076357, "reward_total_mean": 0.3047120273113251, "reward_meter_mean": 0.45798543095588684, "reward_meter_std": 0.24643538892269135, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.9681723713874817, "reward_repeat_soft_std": 0.033738333731889725, "reward_judge_quality_mean": 0.39750000834465027, "reward_judge_quality_std": 0.19031928479671478, "reward_total_composite_mean": 0.3047120273113251, "reward_total_composite_std": 0.26664313673973083} {"timestamp_utc": "2026-04-13T08:33:10Z", "mode": "train", "global_step": 474, "epoch": 0.04761426418884982, "loss": 1.3768, "grad_norm": 20.031064987182617, "learning_rate": 8.566666666666667e-06, "num_tokens": 855436.0, "completions/mean_length": 45.625, "completions/min_length": 16.0, "completions/max_length": 237.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.625, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 237.0, "rewards/meter/mean": 0.7345542907714844, "rewards/meter/std": 0.43264904618263245, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9424101114273071, "rewards/repeat_soft/std": 0.049144160002470016, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.15315140783786774, "rewards/total_composite/mean": 0.4785913825035095, "rewards/total_composite/std": 0.21766385436058044, "reward": 0.4785913825035095, "reward_std": 0.21766385436058044, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17347753047943115, "sampling/sampling_logp_difference/max": 0.9444046020507812, "sampling/importance_sampling_ratio/min": 0.38891109824180603, "sampling/importance_sampling_ratio/mean": 1.0217217206954956, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.500089704990387, "clip_ratio/low_mean": 0.0179412798024714, "clip_ratio/low_min": 0.0179412798024714, "clip_ratio/high_mean": 0.1697557345032692, "clip_ratio/high_max": 0.1697557345032692, "clip_ratio/region_mean": 0.1876970143057406, "reward_total_mean": 0.4785913825035095, "reward_meter_mean": 0.7345542907714844, "reward_meter_std": 0.43264904618263245, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9424101114273071, "reward_repeat_soft_std": 0.049144160002470016, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.15315140783786774, "reward_total_composite_mean": 0.4785913825035095, "reward_total_composite_std": 0.21766385436058044} {"timestamp_utc": "2026-04-13T08:33:16Z", "mode": "train", "global_step": 475, "epoch": 0.047714716223003516, "loss": 0.0029, "grad_norm": 30.126554489135742, "learning_rate": 8.563636363636364e-06, "num_tokens": 856853.0, "completions/mean_length": 32.125, "completions/min_length": 29.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.125, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.5700655579566956, "rewards/meter/std": 0.41669371724128723, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.993262767791748, "rewards/repeat_soft/std": 0.013746356591582298, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.5526953339576721, "rewards/total_composite/std": 0.1694902777671814, "reward": 0.5526953339576721, "reward_std": 0.1694902628660202, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20681506395339966, "sampling/sampling_logp_difference/max": 1.572995662689209, "sampling/importance_sampling_ratio/min": 0.2074228674173355, "sampling/importance_sampling_ratio/mean": 1.050066590309143, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.492697849869728, "clip_ratio/low_mean": 0.11188212595880032, "clip_ratio/low_min": 0.11188212595880032, "clip_ratio/high_mean": 0.08080914337188005, "clip_ratio/high_max": 0.08080914337188005, "clip_ratio/region_mean": 0.19269126933068037, "reward_total_mean": 0.5526953339576721, "reward_meter_mean": 0.5700655579566956, "reward_meter_std": 0.41669371724128723, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.993262767791748, "reward_repeat_soft_std": 0.013746356591582298, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.5526953339576721, "reward_total_composite_std": 0.1694902777671814} {"timestamp_utc": "2026-04-13T08:33:23Z", "mode": "train", "global_step": 476, "epoch": 0.04781516825715721, "loss": -0.0005, "grad_norm": 19.27696418762207, "learning_rate": 8.560606060606062e-06, "num_tokens": 858189.0, "completions/mean_length": 18.0, "completions/min_length": 15.0, "completions/max_length": 24.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 18.0, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 24.0, "rewards/meter/mean": 0.9814795851707458, "rewards/meter/std": 0.011831186711788177, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9000933170318604, "rewards/repeat_soft/std": 0.06073679402470589, "rewards/judge_quality/mean": 0.5687500238418579, "rewards/judge_quality/std": 0.3004015386104584, "rewards/total_composite/mean": 0.6974913477897644, "rewards/total_composite/std": 0.18535955250263214, "reward": 0.6974913477897644, "reward_std": 0.18535956740379333, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18992812931537628, "sampling/sampling_logp_difference/max": 1.2975273132324219, "sampling/importance_sampling_ratio/min": 0.273206502199173, "sampling/importance_sampling_ratio/mean": 1.0402055978775024, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2534800693392754, "clip_ratio/low_mean": 0.08107639150694013, "clip_ratio/low_min": 0.08107639150694013, "clip_ratio/high_mean": 0.04687001742422581, "clip_ratio/high_max": 0.04687001742422581, "clip_ratio/region_mean": 0.12794640893116593, "reward_total_mean": 0.6974913477897644, "reward_meter_mean": 0.9814795851707458, "reward_meter_std": 0.011831186711788177, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9000933170318604, "reward_repeat_soft_std": 0.06073679402470589, "reward_judge_quality_mean": 0.5687500238418579, "reward_judge_quality_std": 0.3004015386104584, "reward_total_composite_mean": 0.6974913477897644, "reward_total_composite_std": 0.18535955250263214} {"timestamp_utc": "2026-04-13T08:33:30Z", "mode": "train", "global_step": 477, "epoch": 0.0479156202913109, "loss": 0.0299, "grad_norm": 18.905426025390625, "learning_rate": 8.557575757575757e-06, "num_tokens": 859789.0, "completions/mean_length": 34.0, "completions/min_length": 29.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.0, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.7084445953369141, "rewards/meter/std": 0.3654259741306305, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9225499629974365, "rewards/repeat_soft/std": 0.11385763436555862, "rewards/judge_quality/mean": 0.6612499952316284, "rewards/judge_quality/std": 0.20883607864379883, "rewards/total_composite/mean": 0.671148419380188, "rewards/total_composite/std": 0.21911805868148804, "reward": 0.671148419380188, "reward_std": 0.21911807358264923, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2158888727426529, "sampling/sampling_logp_difference/max": 1.3481683731079102, "sampling/importance_sampling_ratio/min": 0.2597155272960663, "sampling/importance_sampling_ratio/mean": 0.998117983341217, "sampling/importance_sampling_ratio/max": 1.9984525442123413, "entropy": 1.6007494032382965, "clip_ratio/low_mean": 0.05719418544322252, "clip_ratio/low_min": 0.05719418544322252, "clip_ratio/high_mean": 0.09562562964856625, "clip_ratio/high_max": 0.09562562964856625, "clip_ratio/region_mean": 0.15281981509178877, "reward_total_mean": 0.671148419380188, "reward_meter_mean": 0.7084445953369141, "reward_meter_std": 0.3654259741306305, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9225499629974365, "reward_repeat_soft_std": 0.11385763436555862, "reward_judge_quality_mean": 0.6612499952316284, "reward_judge_quality_std": 0.20883607864379883, "reward_total_composite_mean": 0.671148419380188, "reward_total_composite_std": 0.21911805868148804} {"timestamp_utc": "2026-04-13T08:33:43Z", "mode": "train", "global_step": 478, "epoch": 0.04801607232546459, "loss": -0.0867, "grad_norm": 3.2877233028411865, "learning_rate": 8.554545454545456e-06, "num_tokens": 861386.0, "completions/mean_length": 165.625, "completions/min_length": 44.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 50.16666793823242, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.684908390045166, "rewards/meter/std": 0.3519260287284851, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.15430334210395813, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.9943232536315918, "rewards/repeat_soft/std": 0.0065365200862288475, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.22025957703590393, "rewards/total_composite/mean": 0.34827280044555664, "rewards/total_composite/std": 0.2925608158111572, "reward": 0.34827280044555664, "reward_std": 0.2925608158111572, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23292741179466248, "sampling/sampling_logp_difference/max": 0.9896097183227539, "sampling/importance_sampling_ratio/min": 0.3717217445373535, "sampling/importance_sampling_ratio/mean": 1.0913057327270508, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.8434442281723022, "clip_ratio/low_mean": 0.01785714365541935, "clip_ratio/low_min": 0.01785714365541935, "clip_ratio/high_mean": 0.15885165706276894, "clip_ratio/high_max": 0.15885165706276894, "clip_ratio/region_mean": 0.17670880071818829, "reward_total_mean": 0.34827280044555664, "reward_meter_mean": 0.684908390045166, "reward_meter_std": 0.3519260287284851, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.15430334210395813, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.9943232536315918, "reward_repeat_soft_std": 0.0065365200862288475, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.22025957703590393, "reward_total_composite_mean": 0.34827280044555664, "reward_total_composite_std": 0.2925608158111572} {"timestamp_utc": "2026-04-13T08:33:55Z", "mode": "train", "global_step": 479, "epoch": 0.04811652435961828, "loss": -0.0661, "grad_norm": 3.211437225341797, "learning_rate": 8.551515151515152e-06, "num_tokens": 862782.0, "completions/mean_length": 82.5, "completions/min_length": 16.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 21.142858505249023, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.5668131709098816, "rewards/meter/std": 0.41561824083328247, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.4437500238418579, "rewards/judge_quality/std": 0.23427319526672363, "rewards/total_composite/mean": 0.4611922800540924, "rewards/total_composite/std": 0.21047601103782654, "reward": 0.4611922800540924, "reward_std": 0.21047599613666534, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19073869287967682, "sampling/sampling_logp_difference/max": 0.8275690078735352, "sampling/importance_sampling_ratio/min": 0.43711063265800476, "sampling/importance_sampling_ratio/mean": 1.0617198944091797, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8035916537046432, "clip_ratio/low_mean": 0.08925875462591648, "clip_ratio/low_min": 0.08925875462591648, "clip_ratio/high_mean": 0.06343954289332032, "clip_ratio/high_max": 0.06343954289332032, "clip_ratio/region_mean": 0.1526982975192368, "reward_total_mean": 0.4611922800540924, "reward_meter_mean": 0.5668131709098816, "reward_meter_std": 0.41561824083328247, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.4437500238418579, "reward_judge_quality_std": 0.23427319526672363, "reward_total_composite_mean": 0.4611922800540924, "reward_total_composite_std": 0.21047601103782654} {"timestamp_utc": "2026-04-13T08:34:07Z", "mode": "train", "global_step": 480, "epoch": 0.048216976393771975, "loss": -0.0958, "grad_norm": 1.992726445198059, "learning_rate": 8.548484848484849e-06, "num_tokens": 864478.0, "completions/mean_length": 161.0, "completions/min_length": 36.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 44.0, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.7057309150695801, "rewards/meter/std": 0.42240098118782043, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9811805486679077, "rewards/repeat_soft/std": 0.02527131699025631, "rewards/judge_quality/mean": 0.30000001192092896, "rewards/judge_quality/std": 0.18007934093475342, "rewards/total_composite/mean": 0.41488438844680786, "rewards/total_composite/std": 0.27114206552505493, "reward": 0.41488438844680786, "reward_std": 0.27114206552505493, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1979491263628006, "sampling/sampling_logp_difference/max": 1.4906492233276367, "sampling/importance_sampling_ratio/min": 0.22522638738155365, "sampling/importance_sampling_ratio/mean": 1.0441219806671143, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7134811282157898, "clip_ratio/low_mean": 0.012755102477967739, "clip_ratio/low_min": 0.012755102477967739, "clip_ratio/high_mean": 0.10755876265466213, "clip_ratio/high_max": 0.10755876265466213, "clip_ratio/region_mean": 0.12031386513262987, "reward_total_mean": 0.41488438844680786, "reward_meter_mean": 0.7057309150695801, "reward_meter_std": 0.42240098118782043, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9811805486679077, "reward_repeat_soft_std": 0.02527131699025631, "reward_judge_quality_mean": 0.30000001192092896, "reward_judge_quality_std": 0.18007934093475342, "reward_total_composite_mean": 0.41488438844680786, "reward_total_composite_std": 0.27114206552505493} {"timestamp_utc": "2026-04-13T08:34:18Z", "mode": "train", "global_step": 481, "epoch": 0.04831742842792566, "loss": -0.0458, "grad_norm": 1.5926659107208252, "learning_rate": 8.545454545454546e-06, "num_tokens": 865680.0, "completions/mean_length": 142.25, "completions/min_length": 13.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 19.0, "completions/min_terminated_length": 13.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.7509875893592834, "rewards/meter/std": 0.4451952874660492, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9718749523162842, "rewards/repeat_soft/std": 0.017359137535095215, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.3384418189525604, "rewards/total_composite/mean": 0.5301734209060669, "rewards/total_composite/std": 0.36170145869255066, "reward": 0.5301734209060669, "reward_std": 0.36170145869255066, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17140702903270721, "sampling/sampling_logp_difference/max": 1.5347795486450195, "sampling/importance_sampling_ratio/min": 0.21550320088863373, "sampling/importance_sampling_ratio/mean": 1.048685073852539, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.331137478351593, "clip_ratio/low_mean": 0.028846153989434242, "clip_ratio/low_min": 0.028846153989434242, "clip_ratio/high_mean": 0.08085820684209466, "clip_ratio/high_max": 0.08085820684209466, "clip_ratio/region_mean": 0.1097043608315289, "reward_total_mean": 0.5301734209060669, "reward_meter_mean": 0.7509875893592834, "reward_meter_std": 0.4451952874660492, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9718749523162842, "reward_repeat_soft_std": 0.017359137535095215, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.3384418189525604, "reward_total_composite_mean": 0.5301734209060669, "reward_total_composite_std": 0.36170145869255066} {"timestamp_utc": "2026-04-13T08:34:31Z", "mode": "train", "global_step": 482, "epoch": 0.04841788046207936, "loss": -0.0911, "grad_norm": 4.708141326904297, "learning_rate": 8.542424242424243e-06, "num_tokens": 867433.0, "completions/mean_length": 104.125, "completions/min_length": 40.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 45.85714340209961, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.6382356882095337, "rewards/meter/std": 0.3718057870864868, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9953064918518066, "rewards/repeat_soft/std": 0.006628294009715319, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.22226113080978394, "rewards/total_composite/mean": 0.5219759941101074, "rewards/total_composite/std": 0.25120195746421814, "reward": 0.5219759941101074, "reward_std": 0.25120195746421814, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23645931482315063, "sampling/sampling_logp_difference/max": 1.9882946014404297, "sampling/importance_sampling_ratio/min": 0.1369287371635437, "sampling/importance_sampling_ratio/mean": 1.039041519165039, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8812013119459152, "clip_ratio/low_mean": 0.058726416900753975, "clip_ratio/low_min": 0.058726416900753975, "clip_ratio/high_mean": 0.1265664417296648, "clip_ratio/high_max": 0.1265664417296648, "clip_ratio/region_mean": 0.18529285863041878, "reward_total_mean": 0.5219759941101074, "reward_meter_mean": 0.6382356882095337, "reward_meter_std": 0.3718057870864868, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9953064918518066, "reward_repeat_soft_std": 0.006628294009715319, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.22226113080978394, "reward_total_composite_mean": 0.5219759941101074, "reward_total_composite_std": 0.25120195746421814} {"timestamp_utc": "2026-04-13T08:34:44Z", "mode": "train", "global_step": 483, "epoch": 0.04851833249623305, "loss": -0.0851, "grad_norm": 1.8731340169906616, "learning_rate": 8.539393939393939e-06, "num_tokens": 869103.0, "completions/mean_length": 151.75, "completions/min_length": 29.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 31.666667938232422, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.7622488737106323, "rewards/meter/std": 0.3347495496273041, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9956107139587402, "rewards/repeat_soft/std": 0.007884573191404343, "rewards/judge_quality/mean": 0.3100000023841858, "rewards/judge_quality/std": 0.1718803495168686, "rewards/total_composite/mean": 0.44280362129211426, "rewards/total_composite/std": 0.2757055163383484, "reward": 0.44280362129211426, "reward_std": 0.275705486536026, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22553057968616486, "sampling/sampling_logp_difference/max": 1.406686782836914, "sampling/importance_sampling_ratio/min": 0.24495352804660797, "sampling/importance_sampling_ratio/mean": 1.0452746152877808, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.158455014228821, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.1746977586299181, "clip_ratio/high_max": 0.1746977586299181, "clip_ratio/region_mean": 0.1746977586299181, "reward_total_mean": 0.44280362129211426, "reward_meter_mean": 0.7622488737106323, "reward_meter_std": 0.3347495496273041, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9956107139587402, "reward_repeat_soft_std": 0.007884573191404343, "reward_judge_quality_mean": 0.3100000023841858, "reward_judge_quality_std": 0.1718803495168686, "reward_total_composite_mean": 0.44280362129211426, "reward_total_composite_std": 0.2757055163383484} {"timestamp_utc": "2026-04-13T08:34:55Z", "mode": "train", "global_step": 484, "epoch": 0.04861878453038674, "loss": -0.0574, "grad_norm": 1.668796420097351, "learning_rate": 8.536363636363636e-06, "num_tokens": 870578.0, "completions/mean_length": 275.375, "completions/min_length": 31.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.5, "completions/mean_terminated_length": 38.75, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.6802476048469543, "rewards/meter/std": 0.34469830989837646, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.25877460837364197, "rewards/hard_gate/mean": 0.5, "rewards/hard_gate/std": 0.5345224738121033, "rewards/repeat_soft/mean": 0.9809967875480652, "rewards/repeat_soft/std": 0.05172518268227577, "rewards/judge_quality/mean": 0.4637500047683716, "rewards/judge_quality/std": 0.44580385088920593, "rewards/total_composite/mean": 0.39847850799560547, "rewards/total_composite/std": 0.44206559658050537, "reward": 0.39847850799560547, "reward_std": 0.44206562638282776, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23351459205150604, "sampling/sampling_logp_difference/max": 1.6386666297912598, "sampling/importance_sampling_ratio/min": 0.19423885643482208, "sampling/importance_sampling_ratio/mean": 1.0209335088729858, "sampling/importance_sampling_ratio/max": 1.936795949935913, "entropy": 1.0624631345272064, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.12084288150072098, "clip_ratio/high_max": 0.12084288150072098, "clip_ratio/region_mean": 0.12084288150072098, "reward_total_mean": 0.39847850799560547, "reward_meter_mean": 0.6802476048469543, "reward_meter_std": 0.34469830989837646, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.25877460837364197, "reward_hard_gate_mean": 0.5, "reward_hard_gate_std": 0.5345224738121033, "reward_repeat_soft_mean": 0.9809967875480652, "reward_repeat_soft_std": 0.05172518268227577, "reward_judge_quality_mean": 0.4637500047683716, "reward_judge_quality_std": 0.44580385088920593, "reward_total_composite_mean": 0.39847850799560547, "reward_total_composite_std": 0.44206559658050537} {"timestamp_utc": "2026-04-13T08:35:01Z", "mode": "train", "global_step": 485, "epoch": 0.048719236564540434, "loss": 0.0234, "grad_norm": 26.57079315185547, "learning_rate": 8.533333333333335e-06, "num_tokens": 871947.0, "completions/mean_length": 21.125, "completions/min_length": 18.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.125, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9872746467590332, "rewards/meter/std": 0.006401092279702425, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9247238039970398, "rewards/repeat_soft/std": 0.08560340106487274, "rewards/judge_quality/mean": 0.4024999737739563, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.596958339214325, "rewards/total_composite/std": 0.05244814231991768, "reward": 0.596958339214325, "reward_std": 0.052448149770498276, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24062158167362213, "sampling/sampling_logp_difference/max": 1.3762474060058594, "sampling/importance_sampling_ratio/min": 0.25252440571784973, "sampling/importance_sampling_ratio/mean": 1.1309151649475098, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.681212991476059, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.14498528745025396, "clip_ratio/high_max": 0.14498528745025396, "clip_ratio/region_mean": 0.14498528745025396, "reward_total_mean": 0.596958339214325, "reward_meter_mean": 0.9872746467590332, "reward_meter_std": 0.006401092279702425, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9247238039970398, "reward_repeat_soft_std": 0.08560340106487274, "reward_judge_quality_mean": 0.4024999737739563, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.596958339214325, "reward_total_composite_std": 0.05244814231991768} {"timestamp_utc": "2026-04-13T08:35:07Z", "mode": "train", "global_step": 486, "epoch": 0.04881968859869412, "loss": 0.0409, "grad_norm": 16.259777069091797, "learning_rate": 8.53030303030303e-06, "num_tokens": 873338.0, "completions/mean_length": 21.875, "completions/min_length": 18.0, "completions/max_length": 27.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.875, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.9103868007659912, "rewards/meter/std": 0.2393653243780136, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9541035294532776, "rewards/repeat_soft/std": 0.02374878153204918, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.5367006063461304, "rewards/total_composite/std": 0.07179273664951324, "reward": 0.5367006063461304, "reward_std": 0.07179275155067444, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18916529417037964, "sampling/sampling_logp_difference/max": 1.0688691139221191, "sampling/importance_sampling_ratio/min": 0.3433966338634491, "sampling/importance_sampling_ratio/mean": 1.0399225950241089, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.394902676343918, "clip_ratio/low_mean": 0.12793267704546452, "clip_ratio/low_min": 0.12793267704546452, "clip_ratio/high_mean": 0.08188131637871265, "clip_ratio/high_max": 0.08188131637871265, "clip_ratio/region_mean": 0.20981399342417717, "reward_total_mean": 0.5367006063461304, "reward_meter_mean": 0.9103868007659912, "reward_meter_std": 0.2393653243780136, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9541035294532776, "reward_repeat_soft_std": 0.02374878153204918, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.5367006063461304, "reward_total_composite_std": 0.07179273664951324} {"timestamp_utc": "2026-04-13T08:35:13Z", "mode": "train", "global_step": 487, "epoch": 0.04892014063284782, "loss": 0.0231, "grad_norm": 16.082199096679688, "learning_rate": 8.527272727272728e-06, "num_tokens": 875051.0, "completions/mean_length": 34.125, "completions/min_length": 32.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.125, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.8738746643066406, "rewards/meter/std": 0.23040351271629333, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9863909482955933, "rewards/repeat_soft/std": 0.007166656665503979, "rewards/judge_quality/mean": 0.65625, "rewards/judge_quality/std": 0.13741882145404816, "rewards/total_composite/mean": 0.717002809047699, "rewards/total_composite/std": 0.1237366646528244, "reward": 0.717002809047699, "reward_std": 0.1237366646528244, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18899743258953094, "sampling/sampling_logp_difference/max": 2.695510149002075, "sampling/importance_sampling_ratio/min": 0.06750793755054474, "sampling/importance_sampling_ratio/mean": 1.0138459205627441, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2141579985618591, "clip_ratio/low_mean": 0.08772321604192257, "clip_ratio/low_min": 0.08772321604192257, "clip_ratio/high_mean": 0.13157242350280285, "clip_ratio/high_max": 0.13157242350280285, "clip_ratio/region_mean": 0.21929563954472542, "reward_total_mean": 0.717002809047699, "reward_meter_mean": 0.8738746643066406, "reward_meter_std": 0.23040351271629333, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9863909482955933, "reward_repeat_soft_std": 0.007166656665503979, "reward_judge_quality_mean": 0.65625, "reward_judge_quality_std": 0.13741882145404816, "reward_total_composite_mean": 0.717002809047699, "reward_total_composite_std": 0.1237366646528244} {"timestamp_utc": "2026-04-13T08:35:20Z", "mode": "train", "global_step": 488, "epoch": 0.049020592667001504, "loss": 0.0179, "grad_norm": 8.090012550354004, "learning_rate": 8.524242424242425e-06, "num_tokens": 877412.0, "completions/mean_length": 106.125, "completions/min_length": 91.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.125, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.9861431121826172, "rewards/meter/std": 0.010033386759459972, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7791574597358704, "rewards/repeat_soft/std": 0.14937090873718262, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.5187814235687256, "rewards/total_composite/std": 0.06136050075292587, "reward": 0.5187814235687256, "reward_std": 0.06136050820350647, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15641728043556213, "sampling/sampling_logp_difference/max": 2.371082305908203, "sampling/importance_sampling_ratio/min": 0.09337960928678513, "sampling/importance_sampling_ratio/mean": 1.0398927927017212, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7370524555444717, "clip_ratio/low_mean": 0.038747772574424744, "clip_ratio/low_min": 0.038747772574424744, "clip_ratio/high_mean": 0.10292078740894794, "clip_ratio/high_max": 0.10292078740894794, "clip_ratio/region_mean": 0.1416685599833727, "reward_total_mean": 0.5187814235687256, "reward_meter_mean": 0.9861431121826172, "reward_meter_std": 0.010033386759459972, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7791574597358704, "reward_repeat_soft_std": 0.14937090873718262, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.5187814235687256, "reward_total_composite_std": 0.06136050075292587} {"timestamp_utc": "2026-04-13T08:35:27Z", "mode": "train", "global_step": 489, "epoch": 0.0491210447011552, "loss": -0.0457, "grad_norm": 19.640453338623047, "learning_rate": 8.521212121212123e-06, "num_tokens": 879081.0, "completions/mean_length": 39.625, "completions/min_length": 33.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.625, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.8429540395736694, "rewards/meter/std": 0.17335128784179688, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9792447090148926, "rewards/repeat_soft/std": 0.020537715405225754, "rewards/judge_quality/mean": 0.7150000333786011, "rewards/judge_quality/std": 0.2377273440361023, "rewards/total_composite/mean": 0.6701964139938354, "rewards/total_composite/std": 0.315605491399765, "reward": 0.6701964139938354, "reward_std": 0.315605491399765, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1898537278175354, "sampling/sampling_logp_difference/max": 1.5631322860717773, "sampling/importance_sampling_ratio/min": 0.20947889983654022, "sampling/importance_sampling_ratio/mean": 1.0006773471832275, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.35911126434803, "clip_ratio/low_mean": 0.09288949333131313, "clip_ratio/low_min": 0.09288949333131313, "clip_ratio/high_mean": 0.1005353033542633, "clip_ratio/high_max": 0.1005353033542633, "clip_ratio/region_mean": 0.19342479668557644, "reward_total_mean": 0.6701964139938354, "reward_meter_mean": 0.8429540395736694, "reward_meter_std": 0.17335128784179688, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9792447090148926, "reward_repeat_soft_std": 0.020537715405225754, "reward_judge_quality_mean": 0.7150000333786011, "reward_judge_quality_std": 0.2377273440361023, "reward_total_composite_mean": 0.6701964139938354, "reward_total_composite_std": 0.315605491399765} {"timestamp_utc": "2026-04-13T08:35:33Z", "mode": "train", "global_step": 490, "epoch": 0.04922149673530889, "loss": -0.0337, "grad_norm": 21.66672134399414, "learning_rate": 8.518181818181818e-06, "num_tokens": 880446.0, "completions/mean_length": 22.625, "completions/min_length": 13.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.625, "completions/min_terminated_length": 13.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.7413045167922974, "rewards/meter/std": 0.4184134006500244, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9648010730743408, "rewards/repeat_soft/std": 0.015713289380073547, "rewards/judge_quality/mean": 0.6187499761581421, "rewards/judge_quality/std": 0.24976776540279388, "rewards/total_composite/mean": 0.5909124612808228, "rewards/total_composite/std": 0.2900278866291046, "reward": 0.5909124612808228, "reward_std": 0.2900278568267822, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19013434648513794, "sampling/sampling_logp_difference/max": 0.9521064758300781, "sampling/importance_sampling_ratio/min": 0.40101349353790283, "sampling/importance_sampling_ratio/mean": 1.0587222576141357, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.216610014438629, "clip_ratio/low_mean": 0.04273504298180342, "clip_ratio/low_min": 0.04273504298180342, "clip_ratio/high_mean": 0.16760481614619493, "clip_ratio/high_max": 0.16760481614619493, "clip_ratio/region_mean": 0.21033985912799835, "reward_total_mean": 0.5909124612808228, "reward_meter_mean": 0.7413045167922974, "reward_meter_std": 0.4184134006500244, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9648010730743408, "reward_repeat_soft_std": 0.015713289380073547, "reward_judge_quality_mean": 0.6187499761581421, "reward_judge_quality_std": 0.24976776540279388, "reward_total_composite_mean": 0.5909124612808228, "reward_total_composite_std": 0.2900278866291046} {"timestamp_utc": "2026-04-13T08:35:39Z", "mode": "train", "global_step": 491, "epoch": 0.04932194876946258, "loss": 0.0707, "grad_norm": 14.891180992126465, "learning_rate": 8.515151515151517e-06, "num_tokens": 882310.0, "completions/mean_length": 56.0, "completions/min_length": 48.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.0, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.7073996663093567, "rewards/meter/std": 0.27903494238853455, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.962126612663269, "rewards/repeat_soft/std": 0.033207882195711136, "rewards/judge_quality/mean": 0.4737499952316284, "rewards/judge_quality/std": 0.16291433572769165, "rewards/total_composite/mean": 0.5687705278396606, "rewards/total_composite/std": 0.12742707133293152, "reward": 0.5687705278396606, "reward_std": 0.12742707133293152, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20946232974529266, "sampling/sampling_logp_difference/max": 1.7851076126098633, "sampling/importance_sampling_ratio/min": 0.16777901351451874, "sampling/importance_sampling_ratio/mean": 1.0301470756530762, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7885248810052872, "clip_ratio/low_mean": 0.08910230919718742, "clip_ratio/low_min": 0.08910230919718742, "clip_ratio/high_mean": 0.12386363744735718, "clip_ratio/high_max": 0.12386363744735718, "clip_ratio/region_mean": 0.2129659466445446, "reward_total_mean": 0.5687705278396606, "reward_meter_mean": 0.7073996663093567, "reward_meter_std": 0.27903494238853455, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.962126612663269, "reward_repeat_soft_std": 0.033207882195711136, "reward_judge_quality_mean": 0.4737499952316284, "reward_judge_quality_std": 0.16291433572769165, "reward_total_composite_mean": 0.5687705278396606, "reward_total_composite_std": 0.12742707133293152} {"timestamp_utc": "2026-04-13T08:35:45Z", "mode": "train", "global_step": 492, "epoch": 0.049422400803616276, "loss": -0.0054, "grad_norm": 11.34774112701416, "learning_rate": 8.512121212121213e-06, "num_tokens": 884214.0, "completions/mean_length": 62.0, "completions/min_length": 53.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.0, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.7701433300971985, "rewards/meter/std": 0.35311195254325867, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9936911463737488, "rewards/repeat_soft/std": 0.006063162814825773, "rewards/judge_quality/mean": 0.4699999690055847, "rewards/judge_quality/std": 0.1414213627576828, "rewards/total_composite/mean": 0.5916240811347961, "rewards/total_composite/std": 0.14878517389297485, "reward": 0.5916240811347961, "reward_std": 0.14878515899181366, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20833709836006165, "sampling/sampling_logp_difference/max": 1.5889415740966797, "sampling/importance_sampling_ratio/min": 0.2041415572166443, "sampling/importance_sampling_ratio/mean": 1.0326781272888184, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.634389191865921, "clip_ratio/low_mean": 0.08448945730924606, "clip_ratio/low_min": 0.08448945730924606, "clip_ratio/high_mean": 0.10614589788019657, "clip_ratio/high_max": 0.10614589788019657, "clip_ratio/region_mean": 0.19063535518944263, "reward_total_mean": 0.5916240811347961, "reward_meter_mean": 0.7701433300971985, "reward_meter_std": 0.35311195254325867, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9936911463737488, "reward_repeat_soft_std": 0.006063162814825773, "reward_judge_quality_mean": 0.4699999690055847, "reward_judge_quality_std": 0.1414213627576828, "reward_total_composite_mean": 0.5916240811347961, "reward_total_composite_std": 0.14878517389297485} {"timestamp_utc": "2026-04-13T08:35:55Z", "mode": "train", "global_step": 493, "epoch": 0.049522852837769964, "loss": 0.1141, "grad_norm": 19.325977325439453, "learning_rate": 8.50909090909091e-06, "num_tokens": 885525.0, "completions/mean_length": 19.875, "completions/min_length": 17.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 19.875, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.7870991230010986, "rewards/meter/std": 0.35019874572753906, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.6225000023841858, "rewards/judge_quality/std": 0.24656209349632263, "rewards/total_composite/mean": 0.6802371740341187, "rewards/total_composite/std": 0.21170316636562347, "reward": 0.6802371740341187, "reward_std": 0.21170316636562347, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20156840980052948, "sampling/sampling_logp_difference/max": 1.8572874069213867, "sampling/importance_sampling_ratio/min": 0.1560954749584198, "sampling/importance_sampling_ratio/mean": 1.022932767868042, "sampling/importance_sampling_ratio/max": 1.8618160486221313, "entropy": 1.7631626427173615, "clip_ratio/low_mean": 0.10885781142860651, "clip_ratio/low_min": 0.10885781142860651, "clip_ratio/high_mean": 0.06308049615472555, "clip_ratio/high_max": 0.06308049615472555, "clip_ratio/region_mean": 0.17193830758333206, "reward_total_mean": 0.6802371740341187, "reward_meter_mean": 0.7870991230010986, "reward_meter_std": 0.35019874572753906, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.6225000023841858, "reward_judge_quality_std": 0.24656209349632263, "reward_total_composite_mean": 0.6802371740341187, "reward_total_composite_std": 0.21170316636562347} {"timestamp_utc": "2026-04-13T08:36:06Z", "mode": "train", "global_step": 494, "epoch": 0.04962330487192366, "loss": -0.1293, "grad_norm": 7.374967575073242, "learning_rate": 8.506060606060607e-06, "num_tokens": 887357.0, "completions/mean_length": 128.0, "completions/min_length": 56.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 73.14286041259766, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.7525448799133301, "rewards/meter/std": 0.28220289945602417, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9910486936569214, "rewards/repeat_soft/std": 0.007813452742993832, "rewards/judge_quality/mean": 0.45249998569488525, "rewards/judge_quality/std": 0.21625052392482758, "rewards/total_composite/mean": 0.48917150497436523, "rewards/total_composite/std": 0.31889182329177856, "reward": 0.48917150497436523, "reward_std": 0.31889182329177856, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2065785527229309, "sampling/sampling_logp_difference/max": 2.0177111625671387, "sampling/importance_sampling_ratio/min": 0.13295944035053253, "sampling/importance_sampling_ratio/mean": 0.9862677454948425, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1765784621238708, "clip_ratio/low_mean": 0.04494047723710537, "clip_ratio/low_min": 0.04494047723710537, "clip_ratio/high_mean": 0.11561843566596508, "clip_ratio/high_max": 0.11561843566596508, "clip_ratio/region_mean": 0.16055891290307045, "reward_total_mean": 0.48917150497436523, "reward_meter_mean": 0.7525448799133301, "reward_meter_std": 0.28220289945602417, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9910486936569214, "reward_repeat_soft_std": 0.007813452742993832, "reward_judge_quality_mean": 0.45249998569488525, "reward_judge_quality_std": 0.21625052392482758, "reward_total_composite_mean": 0.48917150497436523, "reward_total_composite_std": 0.31889182329177856} {"timestamp_utc": "2026-04-13T08:36:12Z", "mode": "train", "global_step": 495, "epoch": 0.049723756906077346, "loss": 0.0185, "grad_norm": 17.67508888244629, "learning_rate": 8.503030303030304e-06, "num_tokens": 888939.0, "completions/mean_length": 35.75, "completions/min_length": 30.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.75, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.8715434074401855, "rewards/meter/std": 0.2754010856151581, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9766156077384949, "rewards/repeat_soft/std": 0.023068033158779144, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.6576051115989685, "rewards/total_composite/std": 0.15992027521133423, "reward": 0.6576051115989685, "reward_std": 0.15992026031017303, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15880848467350006, "sampling/sampling_logp_difference/max": 1.0550956726074219, "sampling/importance_sampling_ratio/min": 0.3481591045856476, "sampling/importance_sampling_ratio/mean": 1.0528186559677124, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5566628128290176, "clip_ratio/low_mean": 0.15518195182085037, "clip_ratio/low_min": 0.15518195182085037, "clip_ratio/high_mean": 0.045036764815449715, "clip_ratio/high_max": 0.045036764815449715, "clip_ratio/region_mean": 0.2002187166363001, "reward_total_mean": 0.6576051115989685, "reward_meter_mean": 0.8715434074401855, "reward_meter_std": 0.2754010856151581, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9766156077384949, "reward_repeat_soft_std": 0.023068033158779144, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.6576051115989685, "reward_total_composite_std": 0.15992027521133423} {"timestamp_utc": "2026-04-13T08:36:18Z", "mode": "train", "global_step": 496, "epoch": 0.04982420894023104, "loss": 0.0684, "grad_norm": 20.635723114013672, "learning_rate": 8.5e-06, "num_tokens": 890393.0, "completions/mean_length": 19.75, "completions/min_length": 16.0, "completions/max_length": 23.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 19.75, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 23.0, "rewards/meter/mean": 0.9725742340087891, "rewards/meter/std": 0.040043070912361145, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9284564256668091, "rewards/repeat_soft/std": 0.08202844858169556, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.20860078930854797, "rewards/total_composite/mean": 0.6181299090385437, "rewards/total_composite/std": 0.13773126900196075, "reward": 0.6181299090385437, "reward_std": 0.13773126900196075, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22871235013008118, "sampling/sampling_logp_difference/max": 1.5547971725463867, "sampling/importance_sampling_ratio/min": 0.21123221516609192, "sampling/importance_sampling_ratio/mean": 1.041054129600525, "sampling/importance_sampling_ratio/max": 1.9155250787734985, "entropy": 2.313529849052429, "clip_ratio/low_mean": 0.15082208067178726, "clip_ratio/low_min": 0.15082208067178726, "clip_ratio/high_mean": 0.013888888992369175, "clip_ratio/high_max": 0.013888888992369175, "clip_ratio/region_mean": 0.16471096966415644, "reward_total_mean": 0.6181299090385437, "reward_meter_mean": 0.9725742340087891, "reward_meter_std": 0.040043070912361145, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9284564256668091, "reward_repeat_soft_std": 0.08202844858169556, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.20860078930854797, "reward_total_composite_mean": 0.6181299090385437, "reward_total_composite_std": 0.13773126900196075} {"timestamp_utc": "2026-04-13T08:36:26Z", "mode": "train", "global_step": 497, "epoch": 0.04992466097438473, "loss": 0.4003, "grad_norm": 10.301472663879395, "learning_rate": 8.496969696969697e-06, "num_tokens": 892350.0, "completions/mean_length": 88.625, "completions/min_length": 55.0, "completions/max_length": 251.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 88.625, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 251.0, "rewards/meter/mean": 0.5521746873855591, "rewards/meter/std": 0.4495016932487488, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9963110089302063, "rewards/repeat_soft/std": 0.0030512267258018255, "rewards/judge_quality/mean": 0.4112499952316284, "rewards/judge_quality/std": 0.1797965168952942, "rewards/total_composite/mean": 0.40128767490386963, "rewards/total_composite/std": 0.28263911604881287, "reward": 0.40128767490386963, "reward_std": 0.28263911604881287, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18699179589748383, "sampling/sampling_logp_difference/max": 1.7110190391540527, "sampling/importance_sampling_ratio/min": 0.18068158626556396, "sampling/importance_sampling_ratio/mean": 1.0362626314163208, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.036473363637924, "clip_ratio/low_mean": 0.051028044894337654, "clip_ratio/low_min": 0.051028044894337654, "clip_ratio/high_mean": 0.09508397709578276, "clip_ratio/high_max": 0.09508397709578276, "clip_ratio/region_mean": 0.1461120219901204, "reward_total_mean": 0.40128767490386963, "reward_meter_mean": 0.5521746873855591, "reward_meter_std": 0.4495016932487488, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9963110089302063, "reward_repeat_soft_std": 0.0030512267258018255, "reward_judge_quality_mean": 0.4112499952316284, "reward_judge_quality_std": 0.1797965168952942, "reward_total_composite_mean": 0.40128767490386963, "reward_total_composite_std": 0.28263911604881287} {"timestamp_utc": "2026-04-13T08:36:33Z", "mode": "train", "global_step": 498, "epoch": 0.05002511300853842, "loss": -0.0166, "grad_norm": 19.435779571533203, "learning_rate": 8.493939393939394e-06, "num_tokens": 893804.0, "completions/mean_length": 31.75, "completions/min_length": 26.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.75, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.709601640701294, "rewards/meter/std": 0.4097140431404114, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9933873414993286, "rewards/repeat_soft/std": 0.0044791982509195805, "rewards/judge_quality/mean": 0.5762499570846558, "rewards/judge_quality/std": 0.19167962670326233, "rewards/total_composite/mean": 0.6304831504821777, "rewards/total_composite/std": 0.18362566828727722, "reward": 0.6304831504821777, "reward_std": 0.18362563848495483, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19538196921348572, "sampling/sampling_logp_difference/max": 1.905601978302002, "sampling/importance_sampling_ratio/min": 0.14873307943344116, "sampling/importance_sampling_ratio/mean": 1.0222121477127075, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9627540111541748, "clip_ratio/low_mean": 0.0757354386150837, "clip_ratio/low_min": 0.0757354386150837, "clip_ratio/high_mean": 0.07185769081115723, "clip_ratio/high_max": 0.07185769081115723, "clip_ratio/region_mean": 0.14759312942624092, "reward_total_mean": 0.6304831504821777, "reward_meter_mean": 0.709601640701294, "reward_meter_std": 0.4097140431404114, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9933873414993286, "reward_repeat_soft_std": 0.0044791982509195805, "reward_judge_quality_mean": 0.5762499570846558, "reward_judge_quality_std": 0.19167962670326233, "reward_total_composite_mean": 0.6304831504821777, "reward_total_composite_std": 0.18362566828727722} {"timestamp_utc": "2026-04-13T08:36:39Z", "mode": "train", "global_step": 499, "epoch": 0.05012556504269212, "loss": 0.0002, "grad_norm": 10.80377197265625, "learning_rate": 8.490909090909092e-06, "num_tokens": 895363.0, "completions/mean_length": 47.875, "completions/min_length": 42.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.875, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.8173096179962158, "rewards/meter/std": 0.30853766202926636, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9982814788818359, "rewards/repeat_soft/std": 0.003359787864610553, "rewards/judge_quality/mean": 0.6974999904632568, "rewards/judge_quality/std": 0.2247379571199417, "rewards/total_composite/mean": 0.6352838277816772, "rewards/total_composite/std": 0.31333625316619873, "reward": 0.6352838277816772, "reward_std": 0.31333625316619873, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15956829488277435, "sampling/sampling_logp_difference/max": 2.1714234352111816, "sampling/importance_sampling_ratio/min": 0.11401520669460297, "sampling/importance_sampling_ratio/mean": 1.011419653892517, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1996866837143898, "clip_ratio/low_mean": 0.061936674639582634, "clip_ratio/low_min": 0.061936674639582634, "clip_ratio/high_mean": 0.10531785013154149, "clip_ratio/high_max": 0.10531785013154149, "clip_ratio/region_mean": 0.16725452477112412, "reward_total_mean": 0.6352838277816772, "reward_meter_mean": 0.8173096179962158, "reward_meter_std": 0.30853766202926636, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9982814788818359, "reward_repeat_soft_std": 0.003359787864610553, "reward_judge_quality_mean": 0.6974999904632568, "reward_judge_quality_std": 0.2247379571199417, "reward_total_composite_mean": 0.6352838277816772, "reward_total_composite_std": 0.31333625316619873} {"timestamp_utc": "2026-04-13T08:36:51Z", "mode": "train", "global_step": 500, "epoch": 0.050226017076845805, "loss": -0.1111, "grad_norm": 2.1623306274414062, "learning_rate": 8.487878787878789e-06, "num_tokens": 897051.0, "completions/mean_length": 293.0, "completions/min_length": 72.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.5, "completions/mean_terminated_length": 74.0, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.38441598415374756, "rewards/meter/std": 0.41772931814193726, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.25877460837364197, "rewards/hard_gate/mean": 0.5, "rewards/hard_gate/std": 0.5345224738121033, "rewards/repeat_soft/mean": 0.9976768493652344, "rewards/repeat_soft/std": 0.003024688921868801, "rewards/judge_quality/mean": 0.33500000834465027, "rewards/judge_quality/std": 0.3443005383014679, "rewards/total_composite/mean": 0.32037678360939026, "rewards/total_composite/std": 0.3601384162902832, "reward": 0.32037678360939026, "reward_std": 0.3601383864879608, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1918148249387741, "sampling/sampling_logp_difference/max": 1.6068658828735352, "sampling/importance_sampling_ratio/min": 0.20051506161689758, "sampling/importance_sampling_ratio/mean": 1.045210838317871, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9770064651966095, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.08429248817265034, "clip_ratio/high_max": 0.08429248817265034, "clip_ratio/region_mean": 0.08429248817265034, "reward_total_mean": 0.32037678360939026, "reward_meter_mean": 0.38441598415374756, "reward_meter_std": 0.41772931814193726, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.25877460837364197, "reward_hard_gate_mean": 0.5, "reward_hard_gate_std": 0.5345224738121033, "reward_repeat_soft_mean": 0.9976768493652344, "reward_repeat_soft_std": 0.003024688921868801, "reward_judge_quality_mean": 0.33500000834465027, "reward_judge_quality_std": 0.3443005383014679, "reward_total_composite_mean": 0.32037678360939026, "reward_total_composite_std": 0.3601384162902832} {"timestamp_utc": "2026-04-13T08:37:43Z", "mode": "eval", "global_step": 500, "epoch": 0.050226017076845805, "eval_loss": NaN, "eval_runtime": 52.4101, "eval_samples_per_second": 1.526, "eval_steps_per_second": 0.191, "eval_num_tokens": 897051.0, "eval_completions/mean_length": 73.825, "eval_completions/min_length": 27.2, "eval_completions/max_length": 198.2, "eval_completions/clipped_ratio": 0.025, "eval_completions/mean_terminated_length": 62.648214721679686, "eval_completions/min_terminated_length": 27.2, "eval_completions/max_terminated_length": 119.9, "eval_rewards/meter/mean": 0.6945112824440003, "eval_rewards/meter/std": 0.3352689057588577, "eval_rewards/count_adherence/mean": 0.9570833265781402, "eval_rewards/count_adherence/std": 0.07772159017622471, "eval_rewards/hard_gate/mean": 0.95, "eval_rewards/hard_gate/std": 0.11700168251991272, "eval_rewards/repeat_soft/mean": 0.9513344824314117, "eval_rewards/repeat_soft/std": 0.06777058504521846, "eval_rewards/judge_quality/mean": 0.4666249960660934, "eval_rewards/judge_quality/std": 0.15483680814504625, "eval_rewards/total_composite/mean": 0.5227623641490936, "eval_rewards/total_composite/std": 0.15929170995950698, "eval_reward": 0.5227623641490936, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.1298849441111088, "eval_sampling/sampling_logp_difference/max": 1.1411276817321778, "eval_sampling/importance_sampling_ratio/min": 0.32457238137722016, "eval_sampling/importance_sampling_ratio/mean": 1.0405490040779113, "eval_sampling/importance_sampling_ratio/max": 1.5930978775024414, "eval_entropy": 1.8423508882522583, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5227623641490936, "eval_reward_meter_mean": 0.6945112824440003, "eval_reward_meter_std": 0.3352689057588577, "eval_reward_count_adherence_mean": 0.9570833265781402, "eval_reward_count_adherence_std": 0.07772159017622471, "eval_reward_hard_gate_mean": 0.95, "eval_reward_hard_gate_std": 0.11700168251991272, "eval_reward_repeat_soft_mean": 0.9513344824314117, "eval_reward_repeat_soft_std": 0.06777058504521846, "eval_reward_judge_quality_mean": 0.4666249960660934, "eval_reward_judge_quality_std": 0.15483680814504625, "eval_reward_total_composite_mean": 0.5227623641490936, "eval_reward_total_composite_std": 0.15929170995950698} {"timestamp_utc": "2026-04-13T08:37:52Z", "mode": "train", "global_step": 501, "epoch": 0.0503264691109995, "loss": 0.0167, "grad_norm": 16.002286911010742, "learning_rate": 8.484848484848486e-06, "num_tokens": 898809.0, "completions/mean_length": 50.75, "completions/min_length": 45.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.75, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.7950263023376465, "rewards/meter/std": 0.26878464221954346, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9539934992790222, "rewards/repeat_soft/std": 0.03938141465187073, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.5769858360290527, "rewards/total_composite/std": 0.0846138745546341, "reward": 0.5769858360290527, "reward_std": 0.0846138596534729, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1991541087627411, "sampling/sampling_logp_difference/max": 1.085914134979248, "sampling/importance_sampling_ratio/min": 0.34946075081825256, "sampling/importance_sampling_ratio/mean": 1.047680139541626, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8200058937072754, "clip_ratio/low_mean": 0.043792517855763435, "clip_ratio/low_min": 0.043792517855763435, "clip_ratio/high_mean": 0.12483904603868723, "clip_ratio/high_max": 0.12483904603868723, "clip_ratio/region_mean": 0.16863156389445066, "reward_total_mean": 0.5769858360290527, "reward_meter_mean": 0.7950263023376465, "reward_meter_std": 0.26878464221954346, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9539934992790222, "reward_repeat_soft_std": 0.03938141465187073, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.5769858360290527, "reward_total_composite_std": 0.0846138745546341} {"timestamp_utc": "2026-04-13T08:37:59Z", "mode": "train", "global_step": 502, "epoch": 0.05042692114515319, "loss": 0.0449, "grad_norm": 9.340448379516602, "learning_rate": 8.481818181818182e-06, "num_tokens": 901101.0, "completions/mean_length": 98.5, "completions/min_length": 83.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.5, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.6727849841117859, "rewards/meter/std": 0.3345721662044525, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9954895973205566, "rewards/repeat_soft/std": 0.005467571318149567, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.1524970829486847, "rewards/total_composite/mean": 0.5365538001060486, "rewards/total_composite/std": 0.1377352625131607, "reward": 0.5365538001060486, "reward_std": 0.1377352476119995, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21152883768081665, "sampling/sampling_logp_difference/max": 2.0427350997924805, "sampling/importance_sampling_ratio/min": 0.1296735554933548, "sampling/importance_sampling_ratio/mean": 1.0404530763626099, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.467586889863014, "clip_ratio/low_mean": 0.10863708704710007, "clip_ratio/low_min": 0.10863708704710007, "clip_ratio/high_mean": 0.08720150776207447, "clip_ratio/high_max": 0.08720150776207447, "clip_ratio/region_mean": 0.19583859480917454, "reward_total_mean": 0.5365538001060486, "reward_meter_mean": 0.6727849841117859, "reward_meter_std": 0.3345721662044525, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9954895973205566, "reward_repeat_soft_std": 0.005467571318149567, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.1524970829486847, "reward_total_composite_mean": 0.5365538001060486, "reward_total_composite_std": 0.1377352625131607} {"timestamp_utc": "2026-04-13T08:38:11Z", "mode": "train", "global_step": 503, "epoch": 0.05052737317930688, "loss": -0.0118, "grad_norm": 5.307713985443115, "learning_rate": 8.478787878787879e-06, "num_tokens": 903201.0, "completions/mean_length": 134.5, "completions/min_length": 68.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 80.5714340209961, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 127.0, "rewards/meter/mean": 0.8535789847373962, "rewards/meter/std": 0.24601230025291443, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.18600596487522125, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9460170269012451, "rewards/repeat_soft/std": 0.045325759798288345, "rewards/judge_quality/mean": 0.38999998569488525, "rewards/judge_quality/std": 0.24744409322738647, "rewards/total_composite/mean": 0.45327916741371155, "rewards/total_composite/std": 0.29592910408973694, "reward": 0.45327916741371155, "reward_std": 0.29592907428741455, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2000032216310501, "sampling/sampling_logp_difference/max": 1.9902896881103516, "sampling/importance_sampling_ratio/min": 0.13665582239627838, "sampling/importance_sampling_ratio/mean": 1.0321557521820068, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1861378848552704, "clip_ratio/low_mean": 0.008858268149197102, "clip_ratio/low_min": 0.008858268149197102, "clip_ratio/high_mean": 0.13697792310267687, "clip_ratio/high_max": 0.13697792310267687, "clip_ratio/region_mean": 0.14583619125187397, "reward_total_mean": 0.45327916741371155, "reward_meter_mean": 0.8535789847373962, "reward_meter_std": 0.24601230025291443, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.18600596487522125, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9460170269012451, "reward_repeat_soft_std": 0.045325759798288345, "reward_judge_quality_mean": 0.38999998569488525, "reward_judge_quality_std": 0.24744409322738647, "reward_total_composite_mean": 0.45327916741371155, "reward_total_composite_std": 0.29592910408973694} {"timestamp_utc": "2026-04-13T08:38:22Z", "mode": "train", "global_step": 504, "epoch": 0.05062782521346057, "loss": -0.0667, "grad_norm": 1.7402321100234985, "learning_rate": 8.475757575757576e-06, "num_tokens": 904632.0, "completions/mean_length": 81.875, "completions/min_length": 17.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 20.428571701049805, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.8614223599433899, "rewards/meter/std": 0.3481668531894684, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9605586528778076, "rewards/repeat_soft/std": 0.024460788816213608, "rewards/judge_quality/mean": 0.35999998450279236, "rewards/judge_quality/std": 0.140813946723938, "rewards/total_composite/mean": 0.5268834829330444, "rewards/total_composite/std": 0.21758151054382324, "reward": 0.5268834829330444, "reward_std": 0.21758146584033966, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1638796031475067, "sampling/sampling_logp_difference/max": 0.9651508331298828, "sampling/importance_sampling_ratio/min": 0.3809257447719574, "sampling/importance_sampling_ratio/mean": 1.017634630203247, "sampling/importance_sampling_ratio/max": 1.5614029169082642, "entropy": 1.234602004289627, "clip_ratio/low_mean": 0.004464285913854837, "clip_ratio/low_min": 0.004464285913854837, "clip_ratio/high_mean": 0.09794001141563058, "clip_ratio/high_max": 0.09794001141563058, "clip_ratio/region_mean": 0.10240429732948542, "reward_total_mean": 0.5268834829330444, "reward_meter_mean": 0.8614223599433899, "reward_meter_std": 0.3481668531894684, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9605586528778076, "reward_repeat_soft_std": 0.024460788816213608, "reward_judge_quality_mean": 0.35999998450279236, "reward_judge_quality_std": 0.140813946723938, "reward_total_composite_mean": 0.5268834829330444, "reward_total_composite_std": 0.21758151054382324} {"timestamp_utc": "2026-04-13T08:38:28Z", "mode": "train", "global_step": 505, "epoch": 0.050728277247614265, "loss": 0.1209, "grad_norm": 16.707841873168945, "learning_rate": 8.472727272727274e-06, "num_tokens": 906189.0, "completions/mean_length": 39.625, "completions/min_length": 35.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.625, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.9801276326179504, "rewards/meter/std": 0.013514342717826366, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9287209510803223, "rewards/repeat_soft/std": 0.08914603292942047, "rewards/judge_quality/mean": 0.42499998211860657, "rewards/judge_quality/std": 0.0707106739282608, "rewards/total_composite/mean": 0.6100842952728271, "rewards/total_composite/std": 0.04736431688070297, "reward": 0.6100842952728271, "reward_std": 0.047364309430122375, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21045367419719696, "sampling/sampling_logp_difference/max": 1.5575942993164062, "sampling/importance_sampling_ratio/min": 0.21064220368862152, "sampling/importance_sampling_ratio/mean": 1.0381056070327759, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1052683740854263, "clip_ratio/low_mean": 0.03557233791798353, "clip_ratio/low_min": 0.03557233791798353, "clip_ratio/high_mean": 0.16790770925581455, "clip_ratio/high_max": 0.16790770925581455, "clip_ratio/region_mean": 0.20348004717379808, "reward_total_mean": 0.6100842952728271, "reward_meter_mean": 0.9801276326179504, "reward_meter_std": 0.013514342717826366, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9287209510803223, "reward_repeat_soft_std": 0.08914603292942047, "reward_judge_quality_mean": 0.42499998211860657, "reward_judge_quality_std": 0.0707106739282608, "reward_total_composite_mean": 0.6100842952728271, "reward_total_composite_std": 0.04736431688070297} {"timestamp_utc": "2026-04-13T08:38:34Z", "mode": "train", "global_step": 506, "epoch": 0.05082872928176796, "loss": 0.0806, "grad_norm": 13.86902141571045, "learning_rate": 8.46969696969697e-06, "num_tokens": 907717.0, "completions/mean_length": 41.0, "completions/min_length": 36.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.0, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.6281594038009644, "rewards/meter/std": 0.2802545726299286, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9969707131385803, "rewards/repeat_soft/std": 0.007184124551713467, "rewards/judge_quality/mean": 0.9200000166893005, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7251849174499512, "rewards/total_composite/std": 0.16820569336414337, "reward": 0.7251849174499512, "reward_std": 0.16820570826530457, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11823517829179764, "sampling/sampling_logp_difference/max": 1.3563261032104492, "sampling/importance_sampling_ratio/min": 0.2576054632663727, "sampling/importance_sampling_ratio/mean": 1.0046759843826294, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5806138329207897, "clip_ratio/low_mean": 0.06011569872498512, "clip_ratio/low_min": 0.06011569872498512, "clip_ratio/high_mean": 0.06972717307507992, "clip_ratio/high_max": 0.06972717307507992, "clip_ratio/region_mean": 0.12984287180006504, "reward_total_mean": 0.7251849174499512, "reward_meter_mean": 0.6281594038009644, "reward_meter_std": 0.2802545726299286, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9969707131385803, "reward_repeat_soft_std": 0.007184124551713467, "reward_judge_quality_mean": 0.9200000166893005, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7251849174499512, "reward_total_composite_std": 0.16820569336414337} {"timestamp_utc": "2026-04-13T08:38:45Z", "mode": "train", "global_step": 507, "epoch": 0.05092918131592165, "loss": -0.0453, "grad_norm": 4.813800811767578, "learning_rate": 8.466666666666668e-06, "num_tokens": 909083.0, "completions/mean_length": 86.75, "completions/min_length": 21.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 26.000001907348633, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.8087725639343262, "rewards/meter/std": 0.29176250100135803, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9967178106307983, "rewards/repeat_soft/std": 0.004181791562587023, "rewards/judge_quality/mean": 0.48124998807907104, "rewards/judge_quality/std": 0.3015880584716797, "rewards/total_composite/mean": 0.5315700173377991, "rewards/total_composite/std": 0.3606009781360626, "reward": 0.5315700173377991, "reward_std": 0.36060091853141785, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19166681170463562, "sampling/sampling_logp_difference/max": 0.9522361755371094, "sampling/importance_sampling_ratio/min": 0.38587716221809387, "sampling/importance_sampling_ratio/mean": 1.077354907989502, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2864388674497604, "clip_ratio/low_mean": 0.05902777798473835, "clip_ratio/low_min": 0.05902777798473835, "clip_ratio/high_mean": 0.07762877829372883, "clip_ratio/high_max": 0.07762877829372883, "clip_ratio/region_mean": 0.13665655627846718, "reward_total_mean": 0.5315700173377991, "reward_meter_mean": 0.8087725639343262, "reward_meter_std": 0.29176250100135803, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9967178106307983, "reward_repeat_soft_std": 0.004181791562587023, "reward_judge_quality_mean": 0.48124998807907104, "reward_judge_quality_std": 0.3015880584716797, "reward_total_composite_mean": 0.5315700173377991, "reward_total_composite_std": 0.3606009781360626} {"timestamp_utc": "2026-04-13T08:38:56Z", "mode": "train", "global_step": 508, "epoch": 0.05102963335007534, "loss": -0.0888, "grad_norm": 4.174500942230225, "learning_rate": 8.463636363636364e-06, "num_tokens": 910795.0, "completions/mean_length": 113.0, "completions/min_length": 50.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 56.000003814697266, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.4167751967906952, "rewards/meter/std": 0.3632891774177551, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9934234619140625, "rewards/repeat_soft/std": 0.006900710053741932, "rewards/judge_quality/mean": 0.5362499952316284, "rewards/judge_quality/std": 0.29731839895248413, "rewards/total_composite/mean": 0.4950295090675354, "rewards/total_composite/std": 0.2799857556819916, "reward": 0.4950295090675354, "reward_std": 0.2799857258796692, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20668399333953857, "sampling/sampling_logp_difference/max": 1.6211118698120117, "sampling/importance_sampling_ratio/min": 0.19767877459526062, "sampling/importance_sampling_ratio/mean": 0.9983342289924622, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8299167305231094, "clip_ratio/low_mean": 0.10491517372429371, "clip_ratio/low_min": 0.10491517372429371, "clip_ratio/high_mean": 0.05370689695701003, "clip_ratio/high_max": 0.05370689695701003, "clip_ratio/region_mean": 0.15862207068130374, "reward_total_mean": 0.4950295090675354, "reward_meter_mean": 0.4167751967906952, "reward_meter_std": 0.3632891774177551, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9934234619140625, "reward_repeat_soft_std": 0.006900710053741932, "reward_judge_quality_mean": 0.5362499952316284, "reward_judge_quality_std": 0.29731839895248413, "reward_total_composite_mean": 0.4950295090675354, "reward_total_composite_std": 0.2799857556819916} {"timestamp_utc": "2026-04-13T08:39:02Z", "mode": "train", "global_step": 509, "epoch": 0.05113008538422903, "loss": 0.215, "grad_norm": 16.09877586364746, "learning_rate": 8.460606060606061e-06, "num_tokens": 912375.0, "completions/mean_length": 45.5, "completions/min_length": 36.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.5, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.7594577074050903, "rewards/meter/std": 0.399588018655777, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9873208999633789, "rewards/repeat_soft/std": 0.025526583194732666, "rewards/judge_quality/mean": 0.5362499952316284, "rewards/judge_quality/std": 0.22385823726654053, "rewards/total_composite/mean": 0.6417040824890137, "rewards/total_composite/std": 0.2031925767660141, "reward": 0.6417040824890137, "reward_std": 0.2031925767660141, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1875120997428894, "sampling/sampling_logp_difference/max": 2.4239516258239746, "sampling/importance_sampling_ratio/min": 0.30619949102401733, "sampling/importance_sampling_ratio/mean": 1.038735032081604, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.102756232023239, "clip_ratio/low_mean": 0.12419511005282402, "clip_ratio/low_min": 0.12419511005282402, "clip_ratio/high_mean": 0.070777028799057, "clip_ratio/high_max": 0.070777028799057, "clip_ratio/region_mean": 0.19497213885188103, "reward_total_mean": 0.6417040824890137, "reward_meter_mean": 0.7594577074050903, "reward_meter_std": 0.399588018655777, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9873208999633789, "reward_repeat_soft_std": 0.025526583194732666, "reward_judge_quality_mean": 0.5362499952316284, "reward_judge_quality_std": 0.22385823726654053, "reward_total_composite_mean": 0.6417040824890137, "reward_total_composite_std": 0.2031925767660141} {"timestamp_utc": "2026-04-13T08:39:08Z", "mode": "train", "global_step": 510, "epoch": 0.051230537418382724, "loss": 0.0228, "grad_norm": 16.135772705078125, "learning_rate": 8.457575757575758e-06, "num_tokens": 914171.0, "completions/mean_length": 56.5, "completions/min_length": 49.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.5, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.4396429657936096, "rewards/meter/std": 0.3085101842880249, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9942453503608704, "rewards/repeat_soft/std": 0.00546035123988986, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.4842175245285034, "rewards/total_composite/std": 0.0867922455072403, "reward": 0.4842175245285034, "reward_std": 0.0867922455072403, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22289490699768066, "sampling/sampling_logp_difference/max": 2.671037197113037, "sampling/importance_sampling_ratio/min": 0.0691804364323616, "sampling/importance_sampling_ratio/mean": 1.0352829694747925, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.149636998772621, "clip_ratio/low_mean": 0.0742317233234644, "clip_ratio/low_min": 0.0742317233234644, "clip_ratio/high_mean": 0.10805025603622198, "clip_ratio/high_max": 0.10805025603622198, "clip_ratio/region_mean": 0.18228197935968637, "reward_total_mean": 0.4842175245285034, "reward_meter_mean": 0.4396429657936096, "reward_meter_std": 0.3085101842880249, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9942453503608704, "reward_repeat_soft_std": 0.00546035123988986, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.4842175245285034, "reward_total_composite_std": 0.0867922455072403} {"timestamp_utc": "2026-04-13T08:39:19Z", "mode": "train", "global_step": 511, "epoch": 0.05133098945253641, "loss": -0.1018, "grad_norm": 3.400689125061035, "learning_rate": 8.454545454545455e-06, "num_tokens": 915678.0, "completions/mean_length": 94.375, "completions/min_length": 32.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 34.71428680419922, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.8086185455322266, "rewards/meter/std": 0.3364562392234802, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9483927488327026, "rewards/repeat_soft/std": 0.0429542101919651, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.28965190052986145, "rewards/total_composite/mean": 0.5976336002349854, "rewards/total_composite/std": 0.28357553482055664, "reward": 0.5976336002349854, "reward_std": 0.28357553482055664, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18736304342746735, "sampling/sampling_logp_difference/max": 1.2291431427001953, "sampling/importance_sampling_ratio/min": 0.29254311323165894, "sampling/importance_sampling_ratio/mean": 1.0303345918655396, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3929908722639084, "clip_ratio/low_mean": 0.042258523404598236, "clip_ratio/low_min": 0.042258523404598236, "clip_ratio/high_mean": 0.10468585602939129, "clip_ratio/high_max": 0.10468585602939129, "clip_ratio/region_mean": 0.14694437943398952, "reward_total_mean": 0.5976336002349854, "reward_meter_mean": 0.8086185455322266, "reward_meter_std": 0.3364562392234802, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9483927488327026, "reward_repeat_soft_std": 0.0429542101919651, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.28965190052986145, "reward_total_composite_mean": 0.5976336002349854, "reward_total_composite_std": 0.28357553482055664} {"timestamp_utc": "2026-04-13T08:39:26Z", "mode": "train", "global_step": 512, "epoch": 0.051431441486690106, "loss": 0.0273, "grad_norm": 34.653568267822266, "learning_rate": 8.451515151515151e-06, "num_tokens": 917114.0, "completions/mean_length": 22.5, "completions/min_length": 21.0, "completions/max_length": 24.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.5, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 24.0, "rewards/meter/mean": 0.8293523192405701, "rewards/meter/std": 0.3166838586330414, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9123008251190186, "rewards/repeat_soft/std": 0.15004493296146393, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.5746053457260132, "rewards/total_composite/std": 0.0899825245141983, "reward": 0.5746053457260132, "reward_std": 0.0899825170636177, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1600390523672104, "sampling/sampling_logp_difference/max": 3.9742989540100098, "sampling/importance_sampling_ratio/min": 0.018792470917105675, "sampling/importance_sampling_ratio/mean": 0.9933913350105286, "sampling/importance_sampling_ratio/max": 1.8240867853164673, "entropy": 0.7740676701068878, "clip_ratio/low_mean": 0.05046583851799369, "clip_ratio/low_min": 0.05046583851799369, "clip_ratio/high_mean": 0.0725137647241354, "clip_ratio/high_max": 0.0725137647241354, "clip_ratio/region_mean": 0.12297960324212909, "reward_total_mean": 0.5746053457260132, "reward_meter_mean": 0.8293523192405701, "reward_meter_std": 0.3166838586330414, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9123008251190186, "reward_repeat_soft_std": 0.15004493296146393, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.5746053457260132, "reward_total_composite_std": 0.0899825245141983} {"timestamp_utc": "2026-04-13T08:39:37Z", "mode": "train", "global_step": 513, "epoch": 0.051531893520843794, "loss": -0.1581, "grad_norm": 3.4269769191741943, "learning_rate": 8.44848484848485e-06, "num_tokens": 919291.0, "completions/mean_length": 207.125, "completions/min_length": 94.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 105.5, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.9621840715408325, "rewards/meter/std": 0.035620398819446564, "rewards/count_adherence/mean": 0.675000011920929, "rewards/count_adherence/std": 0.23754701018333435, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.9443340301513672, "rewards/repeat_soft/std": 0.05840669572353363, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.22238561511039734, "rewards/total_composite/mean": 0.36305707693099976, "rewards/total_composite/std": 0.3129875063896179, "reward": 0.36305707693099976, "reward_std": 0.3129875063896179, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1959700733423233, "sampling/sampling_logp_difference/max": 1.6656246185302734, "sampling/importance_sampling_ratio/min": 0.18907251954078674, "sampling/importance_sampling_ratio/mean": 1.042391061782837, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8112002611160278, "clip_ratio/low_mean": 0.026595745235681534, "clip_ratio/low_min": 0.026595745235681534, "clip_ratio/high_mean": 0.10398740135133266, "clip_ratio/high_max": 0.10398740135133266, "clip_ratio/region_mean": 0.1305831465870142, "reward_total_mean": 0.36305707693099976, "reward_meter_mean": 0.9621840715408325, "reward_meter_std": 0.035620398819446564, "reward_count_adherence_mean": 0.675000011920929, "reward_count_adherence_std": 0.23754701018333435, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.9443340301513672, "reward_repeat_soft_std": 0.05840669572353363, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.22238561511039734, "reward_total_composite_mean": 0.36305707693099976, "reward_total_composite_std": 0.3129875063896179} {"timestamp_utc": "2026-04-13T08:39:42Z", "mode": "train", "global_step": 514, "epoch": 0.05163234555499749, "loss": 0.1274, "grad_norm": 28.19949722290039, "learning_rate": 8.445454545454547e-06, "num_tokens": 920785.0, "completions/mean_length": 28.75, "completions/min_length": 24.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.75, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.470637708902359, "rewards/meter/std": 0.3691936433315277, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9973574280738831, "rewards/repeat_soft/std": 0.004874487407505512, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.47808772325515747, "rewards/total_composite/std": 0.10119297355413437, "reward": 0.47808772325515747, "reward_std": 0.10119297355413437, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21324360370635986, "sampling/sampling_logp_difference/max": 2.2050304412841797, "sampling/importance_sampling_ratio/min": 0.1102471649646759, "sampling/importance_sampling_ratio/mean": 1.0146220922470093, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.515910029411316, "clip_ratio/low_mean": 0.06988866627216339, "clip_ratio/low_min": 0.06988866627216339, "clip_ratio/high_mean": 0.13314206153154373, "clip_ratio/high_max": 0.13314206153154373, "clip_ratio/region_mean": 0.20303072780370712, "reward_total_mean": 0.47808772325515747, "reward_meter_mean": 0.470637708902359, "reward_meter_std": 0.3691936433315277, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9973574280738831, "reward_repeat_soft_std": 0.004874487407505512, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.47808772325515747, "reward_total_composite_std": 0.10119297355413437} {"timestamp_utc": "2026-04-13T08:39:48Z", "mode": "train", "global_step": 515, "epoch": 0.05173279758915118, "loss": 0.0781, "grad_norm": 26.827844619750977, "learning_rate": 8.442424242424243e-06, "num_tokens": 922203.0, "completions/mean_length": 26.25, "completions/min_length": 22.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.25, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.518803596496582, "rewards/meter/std": 0.39614447951316833, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9638574719429016, "rewards/repeat_soft/std": 0.049367066472768784, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.48019248247146606, "rewards/total_composite/std": 0.11675118654966354, "reward": 0.48019248247146606, "reward_std": 0.11675118654966354, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2694185972213745, "sampling/sampling_logp_difference/max": 2.053173065185547, "sampling/importance_sampling_ratio/min": 0.12832707166671753, "sampling/importance_sampling_ratio/mean": 1.0390163660049438, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8183883130550385, "clip_ratio/low_mean": 0.08647186495363712, "clip_ratio/low_min": 0.08647186495363712, "clip_ratio/high_mean": 0.11229076609015465, "clip_ratio/high_max": 0.11229076609015465, "clip_ratio/region_mean": 0.19876263104379177, "reward_total_mean": 0.48019248247146606, "reward_meter_mean": 0.518803596496582, "reward_meter_std": 0.39614447951316833, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9638574719429016, "reward_repeat_soft_std": 0.049367066472768784, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.48019248247146606, "reward_total_composite_std": 0.11675118654966354} {"timestamp_utc": "2026-04-13T08:39:59Z", "mode": "train", "global_step": 516, "epoch": 0.05183324962330487, "loss": -0.1341, "grad_norm": 2.6788177490234375, "learning_rate": 8.43939393939394e-06, "num_tokens": 923869.0, "completions/mean_length": 106.25, "completions/min_length": 42.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 48.28571701049805, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.8491782546043396, "rewards/meter/std": 0.22291876375675201, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9589000344276428, "rewards/repeat_soft/std": 0.02780892513692379, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.13274572789669037, "rewards/total_composite/mean": 0.5105176568031311, "rewards/total_composite/std": 0.2141517996788025, "reward": 0.5105176568031311, "reward_std": 0.2141517698764801, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20677360892295837, "sampling/sampling_logp_difference/max": 1.553481101989746, "sampling/importance_sampling_ratio/min": 0.21151040494441986, "sampling/importance_sampling_ratio/mean": 1.0398728847503662, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6627289354801178, "clip_ratio/low_mean": 0.005434782709926367, "clip_ratio/low_min": 0.005434782709926367, "clip_ratio/high_mean": 0.13930488843470812, "clip_ratio/high_max": 0.13930488843470812, "clip_ratio/region_mean": 0.14473967114463449, "reward_total_mean": 0.5105176568031311, "reward_meter_mean": 0.8491782546043396, "reward_meter_std": 0.22291876375675201, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9589000344276428, "reward_repeat_soft_std": 0.02780892513692379, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.13274572789669037, "reward_total_composite_mean": 0.5105176568031311, "reward_total_composite_std": 0.2141517996788025} {"timestamp_utc": "2026-04-13T08:40:10Z", "mode": "train", "global_step": 517, "epoch": 0.051933701657458566, "loss": -0.1557, "grad_norm": 4.7905707359313965, "learning_rate": 8.436363636363637e-06, "num_tokens": 925718.0, "completions/mean_length": 135.125, "completions/min_length": 60.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 81.28572082519531, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 110.0, "rewards/meter/mean": 0.3914533853530884, "rewards/meter/std": 0.37040913105010986, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9313338994979858, "rewards/repeat_soft/std": 0.10688944905996323, "rewards/judge_quality/mean": 0.38999998569488525, "rewards/judge_quality/std": 0.18845234811306, "rewards/total_composite/mean": 0.3853323459625244, "rewards/total_composite/std": 0.18787869811058044, "reward": 0.3853323459625244, "reward_std": 0.18787869811058044, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19800002872943878, "sampling/sampling_logp_difference/max": 1.8448104858398438, "sampling/importance_sampling_ratio/min": 0.15805527567863464, "sampling/importance_sampling_ratio/mean": 1.0412434339523315, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.553493708372116, "clip_ratio/low_mean": 0.06119980476796627, "clip_ratio/low_min": 0.06119980476796627, "clip_ratio/high_mean": 0.08065548166632652, "clip_ratio/high_max": 0.08065548166632652, "clip_ratio/region_mean": 0.1418552864342928, "reward_total_mean": 0.3853323459625244, "reward_meter_mean": 0.3914533853530884, "reward_meter_std": 0.37040913105010986, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9313338994979858, "reward_repeat_soft_std": 0.10688944905996323, "reward_judge_quality_mean": 0.38999998569488525, "reward_judge_quality_std": 0.18845234811306, "reward_total_composite_mean": 0.3853323459625244, "reward_total_composite_std": 0.18787869811058044} {"timestamp_utc": "2026-04-13T08:40:18Z", "mode": "train", "global_step": 518, "epoch": 0.05203415369161225, "loss": 0.3621, "grad_norm": 10.242082595825195, "learning_rate": 8.433333333333334e-06, "num_tokens": 927843.0, "completions/mean_length": 97.625, "completions/min_length": 67.0, "completions/max_length": 206.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.625, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 206.0, "rewards/meter/mean": 0.6179324388504028, "rewards/meter/std": 0.2663954794406891, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.36443448066711426, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8473659753799438, "rewards/repeat_soft/std": 0.1457204520702362, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.48461443185806274, "rewards/total_composite/std": 0.16915346682071686, "reward": 0.48461443185806274, "reward_std": 0.16915346682071686, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14129699766635895, "sampling/sampling_logp_difference/max": 1.7774124145507812, "sampling/importance_sampling_ratio/min": 0.16907508671283722, "sampling/importance_sampling_ratio/mean": 1.019058346748352, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0807285755872726, "clip_ratio/low_mean": 0.057006788440048695, "clip_ratio/low_min": 0.057006788440048695, "clip_ratio/high_mean": 0.11320867296308279, "clip_ratio/high_max": 0.11320867296308279, "clip_ratio/region_mean": 0.17021546140313148, "reward_total_mean": 0.48461443185806274, "reward_meter_mean": 0.6179324388504028, "reward_meter_std": 0.2663954794406891, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.36443448066711426, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8473659753799438, "reward_repeat_soft_std": 0.1457204520702362, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.48461443185806274, "reward_total_composite_std": 0.16915346682071686} {"timestamp_utc": "2026-04-13T08:40:24Z", "mode": "train", "global_step": 519, "epoch": 0.05213460572576595, "loss": 0.0702, "grad_norm": 24.148468017578125, "learning_rate": 8.43030303030303e-06, "num_tokens": 929414.0, "completions/mean_length": 20.375, "completions/min_length": 14.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 20.375, "completions/min_terminated_length": 14.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.8721743822097778, "rewards/meter/std": 0.3301948606967926, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9354029893875122, "rewards/repeat_soft/std": 0.031347643584012985, "rewards/judge_quality/mean": 0.5487500429153442, "rewards/judge_quality/std": 0.22937415540218353, "rewards/total_composite/mean": 0.6613031625747681, "rewards/total_composite/std": 0.18917421996593475, "reward": 0.6613031625747681, "reward_std": 0.18917423486709595, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1783645898103714, "sampling/sampling_logp_difference/max": 1.3344464302062988, "sampling/importance_sampling_ratio/min": 0.2633039057254791, "sampling/importance_sampling_ratio/mean": 0.9993137121200562, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5200989693403244, "clip_ratio/low_mean": 0.1105669941753149, "clip_ratio/low_min": 0.1105669941753149, "clip_ratio/high_mean": 0.057528410106897354, "clip_ratio/high_max": 0.057528410106897354, "clip_ratio/region_mean": 0.16809540428221226, "reward_total_mean": 0.6613031625747681, "reward_meter_mean": 0.8721743822097778, "reward_meter_std": 0.3301948606967926, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9354029893875122, "reward_repeat_soft_std": 0.031347643584012985, "reward_judge_quality_mean": 0.5487500429153442, "reward_judge_quality_std": 0.22937415540218353, "reward_total_composite_mean": 0.6613031625747681, "reward_total_composite_std": 0.18917421996593475} {"timestamp_utc": "2026-04-13T08:40:30Z", "mode": "train", "global_step": 520, "epoch": 0.052235057759919636, "loss": 0.0834, "grad_norm": 20.35475730895996, "learning_rate": 8.427272727272729e-06, "num_tokens": 930876.0, "completions/mean_length": 30.75, "completions/min_length": 27.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.75, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.737522304058075, "rewards/meter/std": 0.14284822344779968, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9978242516517639, "rewards/repeat_soft/std": 0.004353147931396961, "rewards/judge_quality/mean": 0.46875, "rewards/judge_quality/std": 0.11444743722677231, "rewards/total_composite/mean": 0.5755347013473511, "rewards/total_composite/std": 0.07672619819641113, "reward": 0.5755347013473511, "reward_std": 0.07672620564699173, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16212056577205658, "sampling/sampling_logp_difference/max": 1.6107200384140015, "sampling/importance_sampling_ratio/min": 0.19974373281002045, "sampling/importance_sampling_ratio/mean": 1.0059845447540283, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.65384092181921, "clip_ratio/low_mean": 0.09329796768724918, "clip_ratio/low_min": 0.09329796768724918, "clip_ratio/high_mean": 0.03525246400386095, "clip_ratio/high_max": 0.03525246400386095, "clip_ratio/region_mean": 0.12855043169111013, "reward_total_mean": 0.5755347013473511, "reward_meter_mean": 0.737522304058075, "reward_meter_std": 0.14284822344779968, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9978242516517639, "reward_repeat_soft_std": 0.004353147931396961, "reward_judge_quality_mean": 0.46875, "reward_judge_quality_std": 0.11444743722677231, "reward_total_composite_mean": 0.5755347013473511, "reward_total_composite_std": 0.07672619819641113} {"timestamp_utc": "2026-04-13T08:40:36Z", "mode": "train", "global_step": 521, "epoch": 0.05233550979407333, "loss": 0.0146, "grad_norm": 11.828908920288086, "learning_rate": 8.424242424242425e-06, "num_tokens": 932838.0, "completions/mean_length": 67.25, "completions/min_length": 60.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.25, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9194561243057251, "rewards/meter/std": 0.1151917576789856, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8541878461837769, "rewards/repeat_soft/std": 0.06426283717155457, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.5657186508178711, "rewards/total_composite/std": 0.04520780220627785, "reward": 0.5657186508178711, "reward_std": 0.04520779103040695, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1585492193698883, "sampling/sampling_logp_difference/max": 1.5798873901367188, "sampling/importance_sampling_ratio/min": 0.2059982866048813, "sampling/importance_sampling_ratio/mean": 1.0087366104125977, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3772781193256378, "clip_ratio/low_mean": 0.03209841717034578, "clip_ratio/low_min": 0.03209841717034578, "clip_ratio/high_mean": 0.14448866993188858, "clip_ratio/high_max": 0.14448866993188858, "clip_ratio/region_mean": 0.17658708710223436, "reward_total_mean": 0.5657186508178711, "reward_meter_mean": 0.9194561243057251, "reward_meter_std": 0.1151917576789856, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8541878461837769, "reward_repeat_soft_std": 0.06426283717155457, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.5657186508178711, "reward_total_composite_std": 0.04520780220627785} {"timestamp_utc": "2026-04-13T08:40:42Z", "mode": "train", "global_step": 522, "epoch": 0.052435961828227025, "loss": 0.0427, "grad_norm": 13.171037673950195, "learning_rate": 8.421212121212122e-06, "num_tokens": 934492.0, "completions/mean_length": 44.75, "completions/min_length": 40.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.75, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.9922068119049072, "rewards/meter/std": 0.0037105989176779985, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8781986236572266, "rewards/repeat_soft/std": 0.1043878048658371, "rewards/judge_quality/mean": 0.4649999737739563, "rewards/judge_quality/std": 0.1940544992685318, "rewards/total_composite/mean": 0.6316136717796326, "rewards/total_composite/std": 0.13218602538108826, "reward": 0.6316136717796326, "reward_std": 0.13218601047992706, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2231953889131546, "sampling/sampling_logp_difference/max": 1.8242545127868652, "sampling/importance_sampling_ratio/min": 0.16133788228034973, "sampling/importance_sampling_ratio/mean": 1.0236644744873047, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.74095219373703, "clip_ratio/low_mean": 0.13583752140402794, "clip_ratio/low_min": 0.13583752140402794, "clip_ratio/high_mean": 0.06309650093317032, "clip_ratio/high_max": 0.06309650093317032, "clip_ratio/region_mean": 0.19893402233719826, "reward_total_mean": 0.6316136717796326, "reward_meter_mean": 0.9922068119049072, "reward_meter_std": 0.0037105989176779985, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8781986236572266, "reward_repeat_soft_std": 0.1043878048658371, "reward_judge_quality_mean": 0.4649999737739563, "reward_judge_quality_std": 0.1940544992685318, "reward_total_composite_mean": 0.6316136717796326, "reward_total_composite_std": 0.13218602538108826} {"timestamp_utc": "2026-04-13T08:40:49Z", "mode": "train", "global_step": 523, "epoch": 0.05253641386238071, "loss": 0.035, "grad_norm": 22.766992568969727, "learning_rate": 8.418181818181819e-06, "num_tokens": 936371.0, "completions/mean_length": 69.875, "completions/min_length": 53.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.875, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.4473544955253601, "rewards/meter/std": 0.38985496759414673, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9960489869117737, "rewards/repeat_soft/std": 0.001926091150380671, "rewards/judge_quality/mean": 0.5737500190734863, "rewards/judge_quality/std": 0.2672044634819031, "rewards/total_composite/mean": 0.5015914440155029, "rewards/total_composite/std": 0.136424720287323, "reward": 0.5015914440155029, "reward_std": 0.136424720287323, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21991561353206635, "sampling/sampling_logp_difference/max": 2.5158519744873047, "sampling/importance_sampling_ratio/min": 0.08079405128955841, "sampling/importance_sampling_ratio/mean": 0.9984143972396851, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8166926428675652, "clip_ratio/low_mean": 0.09622285235673189, "clip_ratio/low_min": 0.09622285235673189, "clip_ratio/high_mean": 0.07709643803536892, "clip_ratio/high_max": 0.07709643803536892, "clip_ratio/region_mean": 0.1733192903921008, "reward_total_mean": 0.5015914440155029, "reward_meter_mean": 0.4473544955253601, "reward_meter_std": 0.38985496759414673, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9960489869117737, "reward_repeat_soft_std": 0.001926091150380671, "reward_judge_quality_mean": 0.5737500190734863, "reward_judge_quality_std": 0.2672044634819031, "reward_total_composite_mean": 0.5015914440155029, "reward_total_composite_std": 0.136424720287323} {"timestamp_utc": "2026-04-13T08:40:55Z", "mode": "train", "global_step": 524, "epoch": 0.05263686589653441, "loss": 0.0343, "grad_norm": 37.11225128173828, "learning_rate": 8.415151515151516e-06, "num_tokens": 937775.0, "completions/mean_length": 24.5, "completions/min_length": 24.0, "completions/max_length": 25.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.5, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 25.0, "rewards/meter/mean": 0.7918415069580078, "rewards/meter/std": 0.22908450663089752, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.988369345664978, "rewards/repeat_soft/std": 0.008541061542928219, "rewards/judge_quality/mean": 0.8575000166893005, "rewards/judge_quality/std": 0.176776722073555, "rewards/total_composite/mean": 0.7832573652267456, "rewards/total_composite/std": 0.14881549775600433, "reward": 0.7832573652267456, "reward_std": 0.14881551265716553, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11703936755657196, "sampling/sampling_logp_difference/max": 1.1861143112182617, "sampling/importance_sampling_ratio/min": 0.3054056763648987, "sampling/importance_sampling_ratio/mean": 1.0091761350631714, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4233832322061062, "clip_ratio/low_mean": 0.03041666653007269, "clip_ratio/low_min": 0.03041666653007269, "clip_ratio/high_mean": 0.04541666619479656, "clip_ratio/high_max": 0.04541666619479656, "clip_ratio/region_mean": 0.07583333272486925, "reward_total_mean": 0.7832573652267456, "reward_meter_mean": 0.7918415069580078, "reward_meter_std": 0.22908450663089752, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.988369345664978, "reward_repeat_soft_std": 0.008541061542928219, "reward_judge_quality_mean": 0.8575000166893005, "reward_judge_quality_std": 0.176776722073555, "reward_total_composite_mean": 0.7832573652267456, "reward_total_composite_std": 0.14881549775600433} {"timestamp_utc": "2026-04-13T08:41:01Z", "mode": "train", "global_step": 525, "epoch": 0.052737317930688095, "loss": -0.0101, "grad_norm": 21.21489143371582, "learning_rate": 8.412121212121212e-06, "num_tokens": 939377.0, "completions/mean_length": 38.25, "completions/min_length": 31.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.25, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.5493050813674927, "rewards/meter/std": 0.40603959560394287, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9989762306213379, "rewards/repeat_soft/std": 0.0016704278532415628, "rewards/judge_quality/mean": 0.7437499761581421, "rewards/judge_quality/std": 0.2432481348514557, "rewards/total_composite/mean": 0.5870556235313416, "rewards/total_composite/std": 0.1870722472667694, "reward": 0.5870556235313416, "reward_std": 0.18707223236560822, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22771434485912323, "sampling/sampling_logp_difference/max": 1.3903067111968994, "sampling/importance_sampling_ratio/min": 0.2798469662666321, "sampling/importance_sampling_ratio/mean": 1.0537641048431396, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8079886883497238, "clip_ratio/low_mean": 0.13587182201445103, "clip_ratio/low_min": 0.13587182201445103, "clip_ratio/high_mean": 0.13082965649664402, "clip_ratio/high_max": 0.13082965649664402, "clip_ratio/region_mean": 0.26670147851109505, "reward_total_mean": 0.5870556235313416, "reward_meter_mean": 0.5493050813674927, "reward_meter_std": 0.40603959560394287, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9989762306213379, "reward_repeat_soft_std": 0.0016704278532415628, "reward_judge_quality_mean": 0.7437499761581421, "reward_judge_quality_std": 0.2432481348514557, "reward_total_composite_mean": 0.5870556235313416, "reward_total_composite_std": 0.1870722472667694} {"timestamp_utc": "2026-04-13T08:41:07Z", "mode": "train", "global_step": 526, "epoch": 0.05283776996484179, "loss": 0.0285, "grad_norm": 37.78364181518555, "learning_rate": 8.40909090909091e-06, "num_tokens": 940866.0, "completions/mean_length": 28.125, "completions/min_length": 22.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.125, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.6110023260116577, "rewards/meter/std": 0.3921199440956116, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9955050349235535, "rewards/repeat_soft/std": 0.011455883271992207, "rewards/judge_quality/mean": 0.5900000333786011, "rewards/judge_quality/std": 0.23256337642669678, "rewards/total_composite/mean": 0.5320631265640259, "rewards/total_composite/std": 0.3082868456840515, "reward": 0.5320631265640259, "reward_std": 0.3082868456840515, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23938944935798645, "sampling/sampling_logp_difference/max": 2.1711440086364746, "sampling/importance_sampling_ratio/min": 0.11404707282781601, "sampling/importance_sampling_ratio/mean": 0.9849367737770081, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9804778918623924, "clip_ratio/low_mean": 0.07969757076352835, "clip_ratio/low_min": 0.07969757076352835, "clip_ratio/high_mean": 0.10590091813355684, "clip_ratio/high_max": 0.10590091813355684, "clip_ratio/region_mean": 0.1855984888970852, "reward_total_mean": 0.5320631265640259, "reward_meter_mean": 0.6110023260116577, "reward_meter_std": 0.3921199440956116, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9955050349235535, "reward_repeat_soft_std": 0.011455883271992207, "reward_judge_quality_mean": 0.5900000333786011, "reward_judge_quality_std": 0.23256337642669678, "reward_total_composite_mean": 0.5320631265640259, "reward_total_composite_std": 0.3082868456840515} {"timestamp_utc": "2026-04-13T08:41:18Z", "mode": "train", "global_step": 527, "epoch": 0.05293822199899548, "loss": -0.0521, "grad_norm": 13.999894142150879, "learning_rate": 8.406060606060606e-06, "num_tokens": 942629.0, "completions/mean_length": 58.375, "completions/min_length": 45.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.375, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.5185929536819458, "rewards/meter/std": 0.3680461347103119, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9829244613647461, "rewards/repeat_soft/std": 0.03222934156656265, "rewards/judge_quality/mean": 0.5699999928474426, "rewards/judge_quality/std": 0.16035676002502441, "rewards/total_composite/mean": 0.514051079750061, "rewards/total_composite/std": 0.1065848097205162, "reward": 0.514051079750061, "reward_std": 0.10658480226993561, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19776305556297302, "sampling/sampling_logp_difference/max": 2.6942830085754395, "sampling/importance_sampling_ratio/min": 0.06759082525968552, "sampling/importance_sampling_ratio/mean": 1.0025206804275513, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.344873622059822, "clip_ratio/low_mean": 0.09406970255076885, "clip_ratio/low_min": 0.09406970255076885, "clip_ratio/high_mean": 0.10103962756693363, "clip_ratio/high_max": 0.10103962756693363, "clip_ratio/region_mean": 0.19510933011770248, "reward_total_mean": 0.514051079750061, "reward_meter_mean": 0.5185929536819458, "reward_meter_std": 0.3680461347103119, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9829244613647461, "reward_repeat_soft_std": 0.03222934156656265, "reward_judge_quality_mean": 0.5699999928474426, "reward_judge_quality_std": 0.16035676002502441, "reward_total_composite_mean": 0.514051079750061, "reward_total_composite_std": 0.1065848097205162} {"timestamp_utc": "2026-04-13T08:41:27Z", "mode": "train", "global_step": 528, "epoch": 0.05303867403314917, "loss": 0.0579, "grad_norm": 13.73117733001709, "learning_rate": 8.403030303030304e-06, "num_tokens": 944334.0, "completions/mean_length": 55.125, "completions/min_length": 44.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.125, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.8622069358825684, "rewards/meter/std": 0.23428656160831451, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9304995536804199, "rewards/repeat_soft/std": 0.04984993487596512, "rewards/judge_quality/mean": 0.5275000333786011, "rewards/judge_quality/std": 0.18873640894889832, "rewards/total_composite/mean": 0.635810136795044, "rewards/total_composite/std": 0.1618661880493164, "reward": 0.635810136795044, "reward_std": 0.1618662029504776, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19811634719371796, "sampling/sampling_logp_difference/max": 1.4720497131347656, "sampling/importance_sampling_ratio/min": 0.22945468127727509, "sampling/importance_sampling_ratio/mean": 1.0227909088134766, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2130194753408432, "clip_ratio/low_mean": 0.11467535234987736, "clip_ratio/low_min": 0.11467535234987736, "clip_ratio/high_mean": 0.06414141319692135, "clip_ratio/high_max": 0.06414141319692135, "clip_ratio/region_mean": 0.1788167655467987, "reward_total_mean": 0.635810136795044, "reward_meter_mean": 0.8622069358825684, "reward_meter_std": 0.23428656160831451, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9304995536804199, "reward_repeat_soft_std": 0.04984993487596512, "reward_judge_quality_mean": 0.5275000333786011, "reward_judge_quality_std": 0.18873640894889832, "reward_total_composite_mean": 0.635810136795044, "reward_total_composite_std": 0.1618661880493164} {"timestamp_utc": "2026-04-13T08:41:34Z", "mode": "train", "global_step": 529, "epoch": 0.05313912606730286, "loss": 0.1271, "grad_norm": 21.89297103881836, "learning_rate": 8.400000000000001e-06, "num_tokens": 945994.0, "completions/mean_length": 37.5, "completions/min_length": 27.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.5, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.6194216012954712, "rewards/meter/std": 0.3172769546508789, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9978482723236084, "rewards/repeat_soft/std": 0.0031347365584224463, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.22720351815223694, "rewards/total_composite/mean": 0.5457921028137207, "rewards/total_composite/std": 0.07250598818063736, "reward": 0.5457921028137207, "reward_std": 0.07250598073005676, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23090067505836487, "sampling/sampling_logp_difference/max": 2.173288345336914, "sampling/importance_sampling_ratio/min": 0.11380278319120407, "sampling/importance_sampling_ratio/mean": 1.012063980102539, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.971572533249855, "clip_ratio/low_mean": 0.1034488994628191, "clip_ratio/low_min": 0.1034488994628191, "clip_ratio/high_mean": 0.06478441413491964, "clip_ratio/high_max": 0.06478441413491964, "clip_ratio/region_mean": 0.16823331359773874, "reward_total_mean": 0.5457921028137207, "reward_meter_mean": 0.6194216012954712, "reward_meter_std": 0.3172769546508789, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9978482723236084, "reward_repeat_soft_std": 0.0031347365584224463, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.22720351815223694, "reward_total_composite_mean": 0.5457921028137207, "reward_total_composite_std": 0.07250598818063736} {"timestamp_utc": "2026-04-13T08:41:42Z", "mode": "train", "global_step": 530, "epoch": 0.053239578101456554, "loss": 0.06, "grad_norm": 11.848499298095703, "learning_rate": 8.396969696969698e-06, "num_tokens": 947820.0, "completions/mean_length": 60.25, "completions/min_length": 54.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.25, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.7378818988800049, "rewards/meter/std": 0.32036516070365906, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.996192455291748, "rewards/repeat_soft/std": 0.004050884861499071, "rewards/judge_quality/mean": 0.4762499928474426, "rewards/judge_quality/std": 0.09941796213388443, "rewards/total_composite/mean": 0.5040265917778015, "rewards/total_composite/std": 0.23453491926193237, "reward": 0.5040265917778015, "reward_std": 0.23453491926193237, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20574524998664856, "sampling/sampling_logp_difference/max": 2.756950855255127, "sampling/importance_sampling_ratio/min": 0.0634850487112999, "sampling/importance_sampling_ratio/mean": 1.0227644443511963, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5768088102340698, "clip_ratio/low_mean": 0.06443452462553978, "clip_ratio/low_min": 0.06443452462553978, "clip_ratio/high_mean": 0.12275873497128487, "clip_ratio/high_max": 0.12275873497128487, "clip_ratio/region_mean": 0.18719325959682465, "reward_total_mean": 0.5040265917778015, "reward_meter_mean": 0.7378818988800049, "reward_meter_std": 0.32036516070365906, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.996192455291748, "reward_repeat_soft_std": 0.004050884861499071, "reward_judge_quality_mean": 0.4762499928474426, "reward_judge_quality_std": 0.09941796213388443, "reward_total_composite_mean": 0.5040265917778015, "reward_total_composite_std": 0.23453491926193237} {"timestamp_utc": "2026-04-13T08:41:48Z", "mode": "train", "global_step": 531, "epoch": 0.05334003013561025, "loss": -0.0127, "grad_norm": 24.54881477355957, "learning_rate": 8.393939393939394e-06, "num_tokens": 949306.0, "completions/mean_length": 34.75, "completions/min_length": 30.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.75, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.4521898031234741, "rewards/meter/std": 0.4134586751461029, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9783536195755005, "rewards/repeat_soft/std": 0.023981817066669464, "rewards/judge_quality/mean": 0.5112500190734863, "rewards/judge_quality/std": 0.18372632563114166, "rewards/total_composite/mean": 0.5186299085617065, "rewards/total_composite/std": 0.19654682278633118, "reward": 0.5186299085617065, "reward_std": 0.19654680788516998, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22910979390144348, "sampling/sampling_logp_difference/max": 2.3857579231262207, "sampling/importance_sampling_ratio/min": 0.0920192152261734, "sampling/importance_sampling_ratio/mean": 1.021804690361023, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0811530873179436, "clip_ratio/low_mean": 0.12342437170445919, "clip_ratio/low_min": 0.12342437170445919, "clip_ratio/high_mean": 0.09848485141992569, "clip_ratio/high_max": 0.09848485141992569, "clip_ratio/region_mean": 0.22190922312438488, "reward_total_mean": 0.5186299085617065, "reward_meter_mean": 0.4521898031234741, "reward_meter_std": 0.4134586751461029, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9783536195755005, "reward_repeat_soft_std": 0.023981817066669464, "reward_judge_quality_mean": 0.5112500190734863, "reward_judge_quality_std": 0.18372632563114166, "reward_total_composite_mean": 0.5186299085617065, "reward_total_composite_std": 0.19654682278633118} {"timestamp_utc": "2026-04-13T08:41:54Z", "mode": "train", "global_step": 532, "epoch": 0.053440482169763937, "loss": 0.0368, "grad_norm": 21.143571853637695, "learning_rate": 8.390909090909091e-06, "num_tokens": 950966.0, "completions/mean_length": 41.5, "completions/min_length": 37.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.5, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.7355077266693115, "rewards/meter/std": 0.373830646276474, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9728670120239258, "rewards/repeat_soft/std": 0.027256207540631294, "rewards/judge_quality/mean": 0.5237500071525574, "rewards/judge_quality/std": 0.25150617957115173, "rewards/total_composite/mean": 0.535254716873169, "rewards/total_composite/std": 0.2600457966327667, "reward": 0.535254716873169, "reward_std": 0.2600457966327667, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21611906588077545, "sampling/sampling_logp_difference/max": 1.5225722789764404, "sampling/importance_sampling_ratio/min": 0.2181500345468521, "sampling/importance_sampling_ratio/mean": 1.009087085723877, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7905576527118683, "clip_ratio/low_mean": 0.06814480666071177, "clip_ratio/low_min": 0.06814480666071177, "clip_ratio/high_mean": 0.13873954489827156, "clip_ratio/high_max": 0.13873954489827156, "clip_ratio/region_mean": 0.20688435155898333, "reward_total_mean": 0.535254716873169, "reward_meter_mean": 0.7355077266693115, "reward_meter_std": 0.373830646276474, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9728670120239258, "reward_repeat_soft_std": 0.027256207540631294, "reward_judge_quality_mean": 0.5237500071525574, "reward_judge_quality_std": 0.25150617957115173, "reward_total_composite_mean": 0.535254716873169, "reward_total_composite_std": 0.2600457966327667} {"timestamp_utc": "2026-04-13T08:42:00Z", "mode": "train", "global_step": 533, "epoch": 0.05354093420391763, "loss": 0.0203, "grad_norm": 16.195524215698242, "learning_rate": 8.387878787878788e-06, "num_tokens": 952456.0, "completions/mean_length": 33.25, "completions/min_length": 28.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.25, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.9594668745994568, "rewards/meter/std": 0.041511066257953644, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9921087622642517, "rewards/repeat_soft/std": 0.007358473259955645, "rewards/judge_quality/mean": 0.6100000143051147, "rewards/judge_quality/std": 0.19668686389923096, "rewards/total_composite/mean": 0.7268799543380737, "rewards/total_composite/std": 0.11619677394628525, "reward": 0.7268799543380737, "reward_std": 0.11619676649570465, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17709314823150635, "sampling/sampling_logp_difference/max": 4.454348564147949, "sampling/importance_sampling_ratio/min": 0.011627892963588238, "sampling/importance_sampling_ratio/mean": 1.0329965353012085, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2515765875577927, "clip_ratio/low_mean": 0.07639688905328512, "clip_ratio/low_min": 0.07639688905328512, "clip_ratio/high_mean": 0.05112999910488725, "clip_ratio/high_max": 0.05112999910488725, "clip_ratio/region_mean": 0.12752688815817237, "reward_total_mean": 0.7268799543380737, "reward_meter_mean": 0.9594668745994568, "reward_meter_std": 0.041511066257953644, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9921087622642517, "reward_repeat_soft_std": 0.007358473259955645, "reward_judge_quality_mean": 0.6100000143051147, "reward_judge_quality_std": 0.19668686389923096, "reward_total_composite_mean": 0.7268799543380737, "reward_total_composite_std": 0.11619677394628525} {"timestamp_utc": "2026-04-13T08:42:19Z", "mode": "train", "global_step": 534, "epoch": 0.05364138623807132, "loss": -0.0745, "grad_norm": 1.7643216848373413, "learning_rate": 8.384848484848485e-06, "num_tokens": 954003.0, "completions/mean_length": 83.375, "completions/min_length": 17.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 22.142858505249023, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.838414192199707, "rewards/meter/std": 0.3414279520511627, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.13593590259552002, "rewards/total_composite/mean": 0.5166328549385071, "rewards/total_composite/std": 0.21202123165130615, "reward": 0.5166328549385071, "reward_std": 0.21202121675014496, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23563916981220245, "sampling/sampling_logp_difference/max": 1.946211814880371, "sampling/importance_sampling_ratio/min": 0.1428140550851822, "sampling/importance_sampling_ratio/mean": 1.001214623451233, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4632943868637085, "clip_ratio/low_mean": 0.019999999552965164, "clip_ratio/low_min": 0.019999999552965164, "clip_ratio/high_mean": 0.18262881226837635, "clip_ratio/high_max": 0.18262881226837635, "clip_ratio/region_mean": 0.20262881182134151, "reward_total_mean": 0.5166328549385071, "reward_meter_mean": 0.838414192199707, "reward_meter_std": 0.3414279520511627, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.13593590259552002, "reward_total_composite_mean": 0.5166328549385071, "reward_total_composite_std": 0.21202123165130615} {"timestamp_utc": "2026-04-13T08:42:26Z", "mode": "train", "global_step": 535, "epoch": 0.053741838272225013, "loss": 0.0078, "grad_norm": 10.34975528717041, "learning_rate": 8.381818181818183e-06, "num_tokens": 956000.0, "completions/mean_length": 74.625, "completions/min_length": 65.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.625, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.8910391330718994, "rewards/meter/std": 0.240591362118721, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9675438404083252, "rewards/repeat_soft/std": 0.030771438032388687, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.45899564027786255, "rewards/total_composite/std": 0.2833832800388336, "reward": 0.45899564027786255, "reward_std": 0.2833832800388336, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19376224279403687, "sampling/sampling_logp_difference/max": 1.599329948425293, "sampling/importance_sampling_ratio/min": 0.20203185081481934, "sampling/importance_sampling_ratio/mean": 1.0360289812088013, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9693769812583923, "clip_ratio/low_mean": 0.05228365398943424, "clip_ratio/low_min": 0.05228365398943424, "clip_ratio/high_mean": 0.15215030871331692, "clip_ratio/high_max": 0.15215030871331692, "clip_ratio/region_mean": 0.20443396270275116, "reward_total_mean": 0.45899564027786255, "reward_meter_mean": 0.8910391330718994, "reward_meter_std": 0.240591362118721, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9675438404083252, "reward_repeat_soft_std": 0.030771438032388687, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.45899564027786255, "reward_total_composite_std": 0.2833832800388336} {"timestamp_utc": "2026-04-13T08:42:33Z", "mode": "train", "global_step": 536, "epoch": 0.0538422903063787, "loss": 0.058, "grad_norm": 12.88140869140625, "learning_rate": 8.37878787878788e-06, "num_tokens": 958141.0, "completions/mean_length": 86.625, "completions/min_length": 81.0, "completions/max_length": 100.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.625, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 100.0, "rewards/meter/mean": 0.5977858901023865, "rewards/meter/std": 0.26751548051834106, "rewards/count_adherence/mean": 0.8958333134651184, "rewards/count_adherence/std": 0.08625820279121399, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9242637157440186, "rewards/repeat_soft/std": 0.12277748435735703, "rewards/judge_quality/mean": 0.5362499952316284, "rewards/judge_quality/std": 0.21540246903896332, "rewards/total_composite/mean": 0.5444878339767456, "rewards/total_composite/std": 0.1356063038110733, "reward": 0.5444878339767456, "reward_std": 0.1356062889099121, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17371222376823425, "sampling/sampling_logp_difference/max": 3.1495916843414307, "sampling/importance_sampling_ratio/min": 0.042869627475738525, "sampling/importance_sampling_ratio/mean": 0.9901180267333984, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7599944397807121, "clip_ratio/low_mean": 0.05059726629406214, "clip_ratio/low_min": 0.05059726629406214, "clip_ratio/high_mean": 0.08251654729247093, "clip_ratio/high_max": 0.08251654729247093, "clip_ratio/region_mean": 0.13311381358653307, "reward_total_mean": 0.5444878339767456, "reward_meter_mean": 0.5977858901023865, "reward_meter_std": 0.26751548051834106, "reward_count_adherence_mean": 0.8958333134651184, "reward_count_adherence_std": 0.08625820279121399, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9242637157440186, "reward_repeat_soft_std": 0.12277748435735703, "reward_judge_quality_mean": 0.5362499952316284, "reward_judge_quality_std": 0.21540246903896332, "reward_total_composite_mean": 0.5444878339767456, "reward_total_composite_std": 0.1356063038110733} {"timestamp_utc": "2026-04-13T08:42:39Z", "mode": "train", "global_step": 537, "epoch": 0.053942742340532396, "loss": 0.0314, "grad_norm": 20.37936019897461, "learning_rate": 8.375757575757576e-06, "num_tokens": 959775.0, "completions/mean_length": 33.25, "completions/min_length": 26.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.25, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.7705623507499695, "rewards/meter/std": 0.3330211043357849, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9400973320007324, "rewards/repeat_soft/std": 0.04098653420805931, "rewards/judge_quality/mean": 0.6200000047683716, "rewards/judge_quality/std": 0.22677870094776154, "rewards/total_composite/mean": 0.6590745449066162, "rewards/total_composite/std": 0.20637114346027374, "reward": 0.6590745449066162, "reward_std": 0.20637112855911255, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1917899250984192, "sampling/sampling_logp_difference/max": 2.235095977783203, "sampling/importance_sampling_ratio/min": 0.10698185861110687, "sampling/importance_sampling_ratio/mean": 1.0382293462753296, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1163994669914246, "clip_ratio/low_mean": 0.09055245015770197, "clip_ratio/low_min": 0.09055245015770197, "clip_ratio/high_mean": 0.06938631366938353, "clip_ratio/high_max": 0.06938631366938353, "clip_ratio/region_mean": 0.1599387638270855, "reward_total_mean": 0.6590745449066162, "reward_meter_mean": 0.7705623507499695, "reward_meter_std": 0.3330211043357849, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9400973320007324, "reward_repeat_soft_std": 0.04098653420805931, "reward_judge_quality_mean": 0.6200000047683716, "reward_judge_quality_std": 0.22677870094776154, "reward_total_composite_mean": 0.6590745449066162, "reward_total_composite_std": 0.20637114346027374} {"timestamp_utc": "2026-04-13T08:42:45Z", "mode": "train", "global_step": 538, "epoch": 0.05404319437468609, "loss": 0.0148, "grad_norm": 11.401845932006836, "learning_rate": 8.372727272727273e-06, "num_tokens": 961714.0, "completions/mean_length": 65.375, "completions/min_length": 53.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.375, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.9534566402435303, "rewards/meter/std": 0.08393868803977966, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9308726787567139, "rewards/repeat_soft/std": 0.05148974433541298, "rewards/judge_quality/mean": 0.6737500429153442, "rewards/judge_quality/std": 0.17062386870384216, "rewards/total_composite/mean": 0.7569006085395813, "rewards/total_composite/std": 0.11870553344488144, "reward": 0.7569006085395813, "reward_std": 0.11870554834604263, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17962917685508728, "sampling/sampling_logp_difference/max": 1.9633567333221436, "sampling/importance_sampling_ratio/min": 0.14038638770580292, "sampling/importance_sampling_ratio/mean": 1.026145577430725, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2660537511110306, "clip_ratio/low_mean": 0.07088575605303049, "clip_ratio/low_min": 0.07088575605303049, "clip_ratio/high_mean": 0.09831744618713856, "clip_ratio/high_max": 0.09831744618713856, "clip_ratio/region_mean": 0.16920320224016905, "reward_total_mean": 0.7569006085395813, "reward_meter_mean": 0.9534566402435303, "reward_meter_std": 0.08393868803977966, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9308726787567139, "reward_repeat_soft_std": 0.05148974433541298, "reward_judge_quality_mean": 0.6737500429153442, "reward_judge_quality_std": 0.17062386870384216, "reward_total_composite_mean": 0.7569006085395813, "reward_total_composite_std": 0.11870553344488144} {"timestamp_utc": "2026-04-13T08:42:52Z", "mode": "train", "global_step": 539, "epoch": 0.05414364640883978, "loss": -0.0456, "grad_norm": 15.343780517578125, "learning_rate": 8.36969696969697e-06, "num_tokens": 963359.0, "completions/mean_length": 37.625, "completions/min_length": 33.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.625, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.3659902811050415, "rewards/meter/std": 0.3842555284500122, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9993736743927002, "rewards/repeat_soft/std": 0.0009022135054692626, "rewards/judge_quality/mean": 0.8600000143051147, "rewards/judge_quality/std": 0.11122693121433258, "rewards/total_composite/mean": 0.539921760559082, "rewards/total_composite/std": 0.19357290863990784, "reward": 0.539921760559082, "reward_std": 0.19357289373874664, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12317245453596115, "sampling/sampling_logp_difference/max": 1.6021738052368164, "sampling/importance_sampling_ratio/min": 0.20145811140537262, "sampling/importance_sampling_ratio/mean": 1.0202888250350952, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7020215913653374, "clip_ratio/low_mean": 0.09025133214890957, "clip_ratio/low_min": 0.09025133214890957, "clip_ratio/high_mean": 0.038580391090363264, "clip_ratio/high_max": 0.038580391090363264, "clip_ratio/region_mean": 0.12883172323927283, "reward_total_mean": 0.539921760559082, "reward_meter_mean": 0.3659902811050415, "reward_meter_std": 0.3842555284500122, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9993736743927002, "reward_repeat_soft_std": 0.0009022135054692626, "reward_judge_quality_mean": 0.8600000143051147, "reward_judge_quality_std": 0.11122693121433258, "reward_total_composite_mean": 0.539921760559082, "reward_total_composite_std": 0.19357290863990784} {"timestamp_utc": "2026-04-13T08:43:03Z", "mode": "train", "global_step": 540, "epoch": 0.05424409844299347, "loss": -0.0766, "grad_norm": 3.216494083404541, "learning_rate": 8.366666666666667e-06, "num_tokens": 964777.0, "completions/mean_length": 87.25, "completions/min_length": 22.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 26.571430206298828, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.8376508951187134, "rewards/meter/std": 0.30644145607948303, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9671874642372131, "rewards/repeat_soft/std": 0.013258260674774647, "rewards/judge_quality/mean": 0.6862500309944153, "rewards/judge_quality/std": 0.3422170579433441, "rewards/total_composite/mean": 0.7103989124298096, "rewards/total_composite/std": 0.3222620189189911, "reward": 0.7103989124298096, "reward_std": 0.3222619891166687, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13616713881492615, "sampling/sampling_logp_difference/max": 1.0162763595581055, "sampling/importance_sampling_ratio/min": 0.3619401752948761, "sampling/importance_sampling_ratio/mean": 1.0602952241897583, "sampling/importance_sampling_ratio/max": 1.8277971744537354, "entropy": 0.8669724240899086, "clip_ratio/low_mean": 0.04556958540342748, "clip_ratio/low_min": 0.04556958540342748, "clip_ratio/high_mean": 0.045931439846754074, "clip_ratio/high_max": 0.045931439846754074, "clip_ratio/region_mean": 0.09150102525018156, "reward_total_mean": 0.7103989124298096, "reward_meter_mean": 0.8376508951187134, "reward_meter_std": 0.30644145607948303, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9671874642372131, "reward_repeat_soft_std": 0.013258260674774647, "reward_judge_quality_mean": 0.6862500309944153, "reward_judge_quality_std": 0.3422170579433441, "reward_total_composite_mean": 0.7103989124298096, "reward_total_composite_std": 0.3222620189189911} {"timestamp_utc": "2026-04-13T08:43:09Z", "mode": "train", "global_step": 541, "epoch": 0.05434455047714716, "loss": 0.0156, "grad_norm": 14.095969200134277, "learning_rate": 8.363636363636365e-06, "num_tokens": 966520.0, "completions/mean_length": 63.875, "completions/min_length": 50.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.875, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.5517494678497314, "rewards/meter/std": 0.2106204777956009, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9985389709472656, "rewards/repeat_soft/std": 0.0013374953996390104, "rewards/judge_quality/mean": 0.7862499952316284, "rewards/judge_quality/std": 0.21967104077339172, "rewards/total_composite/mean": 0.6565507650375366, "rewards/total_composite/std": 0.15947522222995758, "reward": 0.6565507650375366, "reward_std": 0.15947522222995758, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14598749577999115, "sampling/sampling_logp_difference/max": 2.278921365737915, "sampling/importance_sampling_ratio/min": 0.10239458829164505, "sampling/importance_sampling_ratio/mean": 1.0148061513900757, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7033660002052784, "clip_ratio/low_mean": 0.050267857033759356, "clip_ratio/low_min": 0.050267857033759356, "clip_ratio/high_mean": 0.08420956321060658, "clip_ratio/high_max": 0.08420956321060658, "clip_ratio/region_mean": 0.13447742024436593, "reward_total_mean": 0.6565507650375366, "reward_meter_mean": 0.5517494678497314, "reward_meter_std": 0.2106204777956009, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9985389709472656, "reward_repeat_soft_std": 0.0013374953996390104, "reward_judge_quality_mean": 0.7862499952316284, "reward_judge_quality_std": 0.21967104077339172, "reward_total_composite_mean": 0.6565507650375366, "reward_total_composite_std": 0.15947522222995758} {"timestamp_utc": "2026-04-13T08:43:15Z", "mode": "train", "global_step": 542, "epoch": 0.054445002511300855, "loss": 0.0043, "grad_norm": 17.3575439453125, "learning_rate": 8.360606060606062e-06, "num_tokens": 968089.0, "completions/mean_length": 33.125, "completions/min_length": 28.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.125, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.8381178379058838, "rewards/meter/std": 0.2982902526855469, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9539903998374939, "rewards/repeat_soft/std": 0.06638079881668091, "rewards/judge_quality/mean": 0.5824999809265137, "rewards/judge_quality/std": 0.23260943591594696, "rewards/total_composite/mean": 0.67268967628479, "rewards/total_composite/std": 0.18456526100635529, "reward": 0.67268967628479, "reward_std": 0.1845652461051941, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1786205917596817, "sampling/sampling_logp_difference/max": 1.7740802764892578, "sampling/importance_sampling_ratio/min": 0.16963939368724823, "sampling/importance_sampling_ratio/mean": 1.026566505432129, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6051879823207855, "clip_ratio/low_mean": 0.11004703678190708, "clip_ratio/low_min": 0.11004703678190708, "clip_ratio/high_mean": 0.08839285932481289, "clip_ratio/high_max": 0.08839285932481289, "clip_ratio/region_mean": 0.19843989610671997, "reward_total_mean": 0.67268967628479, "reward_meter_mean": 0.8381178379058838, "reward_meter_std": 0.2982902526855469, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9539903998374939, "reward_repeat_soft_std": 0.06638079881668091, "reward_judge_quality_mean": 0.5824999809265137, "reward_judge_quality_std": 0.23260943591594696, "reward_total_composite_mean": 0.67268967628479, "reward_total_composite_std": 0.18456526100635529} {"timestamp_utc": "2026-04-13T08:43:26Z", "mode": "train", "global_step": 543, "epoch": 0.05454545454545454, "loss": -0.0796, "grad_norm": 3.910327434539795, "learning_rate": 8.357575757575759e-06, "num_tokens": 969548.0, "completions/mean_length": 88.375, "completions/min_length": 22.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 27.85714340209961, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.45423436164855957, "rewards/meter/std": 0.38455843925476074, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9979463815689087, "rewards/repeat_soft/std": 0.005131691228598356, "rewards/judge_quality/mean": 0.36374998092651367, "rewards/judge_quality/std": 0.14302222430706024, "rewards/total_composite/mean": 0.4281260073184967, "rewards/total_composite/std": 0.20449328422546387, "reward": 0.4281260073184967, "reward_std": 0.20449328422546387, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20579834282398224, "sampling/sampling_logp_difference/max": 1.9135963916778564, "sampling/importance_sampling_ratio/min": 0.14754877984523773, "sampling/importance_sampling_ratio/mean": 1.0137248039245605, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2950005009770393, "clip_ratio/low_mean": 0.06727272644639015, "clip_ratio/low_min": 0.06727272644639015, "clip_ratio/high_mean": 0.09855741448700428, "clip_ratio/high_max": 0.09855741448700428, "clip_ratio/region_mean": 0.16583014093339443, "reward_total_mean": 0.4281260073184967, "reward_meter_mean": 0.45423436164855957, "reward_meter_std": 0.38455843925476074, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9979463815689087, "reward_repeat_soft_std": 0.005131691228598356, "reward_judge_quality_mean": 0.36374998092651367, "reward_judge_quality_std": 0.14302222430706024, "reward_total_composite_mean": 0.4281260073184967, "reward_total_composite_std": 0.20449328422546387} {"timestamp_utc": "2026-04-13T08:43:32Z", "mode": "train", "global_step": 544, "epoch": 0.05464590657960824, "loss": 0.0019, "grad_norm": 27.745807647705078, "learning_rate": 8.354545454545455e-06, "num_tokens": 971066.0, "completions/mean_length": 25.75, "completions/min_length": 20.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.75, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.7400469183921814, "rewards/meter/std": 0.35674533247947693, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9558712244033813, "rewards/repeat_soft/std": 0.018749024718999863, "rewards/judge_quality/mean": 0.6150000095367432, "rewards/judge_quality/std": 0.25286927819252014, "rewards/total_composite/mean": 0.6075124740600586, "rewards/total_composite/std": 0.15838244557380676, "reward": 0.6075124740600586, "reward_std": 0.15838246047496796, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1736089140176773, "sampling/sampling_logp_difference/max": 1.2797205448150635, "sampling/importance_sampling_ratio/min": 0.3145306706428528, "sampling/importance_sampling_ratio/mean": 1.0237181186676025, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1577944606542587, "clip_ratio/low_mean": 0.06673119682818651, "clip_ratio/low_min": 0.06673119682818651, "clip_ratio/high_mean": 0.05760630592703819, "clip_ratio/high_max": 0.05760630592703819, "clip_ratio/region_mean": 0.1243375027552247, "reward_total_mean": 0.6075124740600586, "reward_meter_mean": 0.7400469183921814, "reward_meter_std": 0.35674533247947693, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9558712244033813, "reward_repeat_soft_std": 0.018749024718999863, "reward_judge_quality_mean": 0.6150000095367432, "reward_judge_quality_std": 0.25286927819252014, "reward_total_composite_mean": 0.6075124740600586, "reward_total_composite_std": 0.15838244557380676} {"timestamp_utc": "2026-04-13T08:43:38Z", "mode": "train", "global_step": 545, "epoch": 0.05474635861376193, "loss": 0.0132, "grad_norm": 14.433030128479004, "learning_rate": 8.351515151515152e-06, "num_tokens": 972727.0, "completions/mean_length": 43.625, "completions/min_length": 40.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.625, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.7088823318481445, "rewards/meter/std": 0.25771865248680115, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.996931791305542, "rewards/repeat_soft/std": 0.003609251230955124, "rewards/judge_quality/mean": 0.8362500667572021, "rewards/judge_quality/std": 0.17104199528694153, "rewards/total_composite/mean": 0.7405744791030884, "rewards/total_composite/std": 0.18008670210838318, "reward": 0.7405744791030884, "reward_std": 0.18008668720722198, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1720448136329651, "sampling/sampling_logp_difference/max": 1.806600570678711, "sampling/importance_sampling_ratio/min": 0.1642114222049713, "sampling/importance_sampling_ratio/mean": 1.0251959562301636, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3306350335478783, "clip_ratio/low_mean": 0.05352247226983309, "clip_ratio/low_min": 0.05352247226983309, "clip_ratio/high_mean": 0.10940993577241898, "clip_ratio/high_max": 0.10940993577241898, "clip_ratio/region_mean": 0.16293240804225206, "reward_total_mean": 0.7405744791030884, "reward_meter_mean": 0.7088823318481445, "reward_meter_std": 0.25771865248680115, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.996931791305542, "reward_repeat_soft_std": 0.003609251230955124, "reward_judge_quality_mean": 0.8362500667572021, "reward_judge_quality_std": 0.17104199528694153, "reward_total_composite_mean": 0.7405744791030884, "reward_total_composite_std": 0.18008670210838318} {"timestamp_utc": "2026-04-13T08:43:44Z", "mode": "train", "global_step": 546, "epoch": 0.05484681064791562, "loss": 0.0764, "grad_norm": 22.39914321899414, "learning_rate": 8.348484848484849e-06, "num_tokens": 974133.0, "completions/mean_length": 24.75, "completions/min_length": 20.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.75, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.9132776856422424, "rewards/meter/std": 0.14427873492240906, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.5974857807159424, "rewards/total_composite/std": 0.037712667137384415, "reward": 0.5974857807159424, "reward_std": 0.03771267458796501, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13518214225769043, "sampling/sampling_logp_difference/max": 0.8918828964233398, "sampling/importance_sampling_ratio/min": 0.4098832607269287, "sampling/importance_sampling_ratio/mean": 1.0276554822921753, "sampling/importance_sampling_ratio/max": 1.8256620168685913, "entropy": 1.02891393750906, "clip_ratio/low_mean": 0.013392857741564512, "clip_ratio/low_min": 0.013392857741564512, "clip_ratio/high_mean": 0.10959415882825851, "clip_ratio/high_max": 0.10959415882825851, "clip_ratio/region_mean": 0.12298701656982303, "reward_total_mean": 0.5974857807159424, "reward_meter_mean": 0.9132776856422424, "reward_meter_std": 0.14427873492240906, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.5974857807159424, "reward_total_composite_std": 0.037712667137384415} {"timestamp_utc": "2026-04-13T08:43:50Z", "mode": "train", "global_step": 547, "epoch": 0.054947262682069314, "loss": 0.0325, "grad_norm": 15.003937721252441, "learning_rate": 8.345454545454546e-06, "num_tokens": 975818.0, "completions/mean_length": 33.625, "completions/min_length": 26.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.625, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.9554383158683777, "rewards/meter/std": 0.062374863773584366, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9542199969291687, "rewards/repeat_soft/std": 0.04472750797867775, "rewards/judge_quality/mean": 0.5974999666213989, "rewards/judge_quality/std": 0.22663691639900208, "rewards/total_composite/mean": 0.6425566673278809, "rewards/total_composite/std": 0.2965072989463806, "reward": 0.6425566673278809, "reward_std": 0.2965072989463806, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16646623611450195, "sampling/sampling_logp_difference/max": 1.8456273078918457, "sampling/importance_sampling_ratio/min": 0.15792621672153473, "sampling/importance_sampling_ratio/mean": 1.013169288635254, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1770803853869438, "clip_ratio/low_mean": 0.12836424820125103, "clip_ratio/low_min": 0.12836424820125103, "clip_ratio/high_mean": 0.06548554822802544, "clip_ratio/high_max": 0.06548554822802544, "clip_ratio/region_mean": 0.19384979642927647, "reward_total_mean": 0.6425566673278809, "reward_meter_mean": 0.9554383158683777, "reward_meter_std": 0.062374863773584366, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9542199969291687, "reward_repeat_soft_std": 0.04472750797867775, "reward_judge_quality_mean": 0.5974999666213989, "reward_judge_quality_std": 0.22663691639900208, "reward_total_composite_mean": 0.6425566673278809, "reward_total_composite_std": 0.2965072989463806} {"timestamp_utc": "2026-04-13T08:43:56Z", "mode": "train", "global_step": 548, "epoch": 0.055047714716223, "loss": 0.0166, "grad_norm": 18.18979835510254, "learning_rate": 8.342424242424244e-06, "num_tokens": 977363.0, "completions/mean_length": 35.125, "completions/min_length": 33.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.125, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.9609639644622803, "rewards/meter/std": 0.060940779745578766, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9969215393066406, "rewards/repeat_soft/std": 0.005694178864359856, "rewards/judge_quality/mean": 0.8575000166893005, "rewards/judge_quality/std": 0.176776722073555, "rewards/total_composite/mean": 0.8910194635391235, "rewards/total_composite/std": 0.12894363701343536, "reward": 0.8910194635391235, "reward_std": 0.12894362211227417, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16594471037387848, "sampling/sampling_logp_difference/max": 1.630638599395752, "sampling/importance_sampling_ratio/min": 0.19580449163913727, "sampling/importance_sampling_ratio/mean": 1.0251747369766235, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8938365876674652, "clip_ratio/low_mean": 0.025735294446349144, "clip_ratio/low_min": 0.025735294446349144, "clip_ratio/high_mean": 0.07592684868723154, "clip_ratio/high_max": 0.07592684868723154, "clip_ratio/region_mean": 0.10166214313358068, "reward_total_mean": 0.8910194635391235, "reward_meter_mean": 0.9609639644622803, "reward_meter_std": 0.060940779745578766, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9969215393066406, "reward_repeat_soft_std": 0.005694178864359856, "reward_judge_quality_mean": 0.8575000166893005, "reward_judge_quality_std": 0.176776722073555, "reward_total_composite_mean": 0.8910194635391235, "reward_total_composite_std": 0.12894363701343536} {"timestamp_utc": "2026-04-13T08:44:03Z", "mode": "train", "global_step": 549, "epoch": 0.0551481667503767, "loss": 0.0707, "grad_norm": 17.993812561035156, "learning_rate": 8.339393939393941e-06, "num_tokens": 979548.0, "completions/mean_length": 89.125, "completions/min_length": 61.0, "completions/max_length": 147.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.125, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 147.0, "rewards/meter/mean": 0.5475062727928162, "rewards/meter/std": 0.3193122446537018, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9620498418807983, "rewards/repeat_soft/std": 0.03260951116681099, "rewards/judge_quality/mean": 0.38999998569488525, "rewards/judge_quality/std": 0.18071287870407104, "rewards/total_composite/mean": 0.4460676908493042, "rewards/total_composite/std": 0.2205701619386673, "reward": 0.4460676908493042, "reward_std": 0.2205701470375061, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1919461488723755, "sampling/sampling_logp_difference/max": 2.2864503860473633, "sampling/importance_sampling_ratio/min": 0.10162656009197235, "sampling/importance_sampling_ratio/mean": 1.0069726705551147, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2636125013232231, "clip_ratio/low_mean": 0.0745668588206172, "clip_ratio/low_min": 0.0745668588206172, "clip_ratio/high_mean": 0.10123595409095287, "clip_ratio/high_max": 0.10123595409095287, "clip_ratio/region_mean": 0.17580281291157007, "reward_total_mean": 0.4460676908493042, "reward_meter_mean": 0.5475062727928162, "reward_meter_std": 0.3193122446537018, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9620498418807983, "reward_repeat_soft_std": 0.03260951116681099, "reward_judge_quality_mean": 0.38999998569488525, "reward_judge_quality_std": 0.18071287870407104, "reward_total_composite_mean": 0.4460676908493042, "reward_total_composite_std": 0.2205701619386673} {"timestamp_utc": "2026-04-13T08:44:10Z", "mode": "train", "global_step": 550, "epoch": 0.055248618784530384, "loss": -0.0036, "grad_norm": 10.08622932434082, "learning_rate": 8.336363636363636e-06, "num_tokens": 981377.0, "completions/mean_length": 59.625, "completions/min_length": 51.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.625, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9905789494514465, "rewards/meter/std": 0.004395305644720793, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7257529497146606, "rewards/repeat_soft/std": 0.16920576989650726, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.10507651418447495, "rewards/total_composite/mean": 0.6059386134147644, "rewards/total_composite/std": 0.07042774558067322, "reward": 0.6059386134147644, "reward_std": 0.07042776048183441, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16200944781303406, "sampling/sampling_logp_difference/max": 2.120972156524658, "sampling/importance_sampling_ratio/min": 0.11991500109434128, "sampling/importance_sampling_ratio/mean": 1.020952820777893, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2838521227240562, "clip_ratio/low_mean": 0.056965377647429705, "clip_ratio/low_min": 0.056965377647429705, "clip_ratio/high_mean": 0.09082790184766054, "clip_ratio/high_max": 0.09082790184766054, "clip_ratio/region_mean": 0.14779327949509025, "reward_total_mean": 0.6059386134147644, "reward_meter_mean": 0.9905789494514465, "reward_meter_std": 0.004395305644720793, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7257529497146606, "reward_repeat_soft_std": 0.16920576989650726, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.10507651418447495, "reward_total_composite_mean": 0.6059386134147644, "reward_total_composite_std": 0.07042774558067322} {"timestamp_utc": "2026-04-13T08:44:53Z", "mode": "eval", "global_step": 550, "epoch": 0.055248618784530384, "eval_loss": NaN, "eval_runtime": 43.6753, "eval_samples_per_second": 1.832, "eval_steps_per_second": 0.229, "eval_num_tokens": 981377.0, "eval_completions/mean_length": 71.7875, "eval_completions/min_length": 26.6, "eval_completions/max_length": 143.4, "eval_completions/clipped_ratio": 0.025, "eval_completions/mean_terminated_length": 60.645833587646486, "eval_completions/min_terminated_length": 26.6, "eval_completions/max_terminated_length": 102.9, "eval_rewards/meter/mean": 0.6610803008079529, "eval_rewards/meter/std": 0.33115275800228117, "eval_rewards/count_adherence/mean": 0.9912499904632568, "eval_rewards/count_adherence/std": 0.02474873922765255, "eval_rewards/hard_gate/mean": 0.9125, "eval_rewards/hard_gate/std": 0.19317627549171448, "eval_rewards/repeat_soft/mean": 0.9651135087013245, "eval_rewards/repeat_soft/std": 0.03794628819450736, "eval_rewards/judge_quality/mean": 0.5327500224113464, "eval_rewards/judge_quality/std": 0.20415999740362167, "eval_rewards/total_composite/mean": 0.5292426437139511, "eval_rewards/total_composite/std": 0.19960086941719055, "eval_reward": 0.5292426437139511, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.10629716143012047, "eval_sampling/sampling_logp_difference/max": 1.1720905303955078, "eval_sampling/importance_sampling_ratio/min": 0.3158396989107132, "eval_sampling/importance_sampling_ratio/mean": 1.023800778388977, "eval_sampling/importance_sampling_ratio/max": 1.4582764148712157, "eval_entropy": 1.327412235736847, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5292426437139511, "eval_reward_meter_mean": 0.6610803008079529, "eval_reward_meter_std": 0.33115275800228117, "eval_reward_count_adherence_mean": 0.9912499904632568, "eval_reward_count_adherence_std": 0.02474873922765255, "eval_reward_hard_gate_mean": 0.9125, "eval_reward_hard_gate_std": 0.19317627549171448, "eval_reward_repeat_soft_mean": 0.9651135087013245, "eval_reward_repeat_soft_std": 0.03794628819450736, "eval_reward_judge_quality_mean": 0.5327500224113464, "eval_reward_judge_quality_std": 0.20415999740362167, "eval_reward_total_composite_mean": 0.5292426437139511, "eval_reward_total_composite_std": 0.19960086941719055} {"timestamp_utc": "2026-04-13T08:45:04Z", "mode": "train", "global_step": 551, "epoch": 0.05534907081868408, "loss": 0.1879, "grad_norm": 20.56439971923828, "learning_rate": 8.333333333333334e-06, "num_tokens": 982994.0, "completions/mean_length": 40.125, "completions/min_length": 31.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.125, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.6993616223335266, "rewards/meter/std": 0.37903064489364624, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9672762155532837, "rewards/repeat_soft/std": 0.0380360409617424, "rewards/judge_quality/mean": 0.5349999666213989, "rewards/judge_quality/std": 0.18431341648101807, "rewards/total_composite/mean": 0.5190733671188354, "rewards/total_composite/std": 0.21756958961486816, "reward": 0.5190733671188354, "reward_std": 0.21756958961486816, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20431751012802124, "sampling/sampling_logp_difference/max": 1.978684425354004, "sampling/importance_sampling_ratio/min": 0.13825100660324097, "sampling/importance_sampling_ratio/mean": 1.0076302289962769, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4213807880878448, "clip_ratio/low_mean": 0.044801464304327965, "clip_ratio/low_min": 0.044801464304327965, "clip_ratio/high_mean": 0.12536935694515705, "clip_ratio/high_max": 0.12536935694515705, "clip_ratio/region_mean": 0.17017082124948502, "reward_total_mean": 0.5190733671188354, "reward_meter_mean": 0.6993616223335266, "reward_meter_std": 0.37903064489364624, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9672762155532837, "reward_repeat_soft_std": 0.0380360409617424, "reward_judge_quality_mean": 0.5349999666213989, "reward_judge_quality_std": 0.18431341648101807, "reward_total_composite_mean": 0.5190733671188354, "reward_total_composite_std": 0.21756958961486816} {"timestamp_utc": "2026-04-13T08:45:10Z", "mode": "train", "global_step": 552, "epoch": 0.05544952285283777, "loss": -0.0518, "grad_norm": 18.777772903442383, "learning_rate": 8.330303030303031e-06, "num_tokens": 984750.0, "completions/mean_length": 39.5, "completions/min_length": 32.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.5, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.8659785389900208, "rewards/meter/std": 0.17113295197486877, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9889838099479675, "rewards/repeat_soft/std": 0.008318226784467697, "rewards/judge_quality/mean": 0.4937500059604645, "rewards/judge_quality/std": 0.1728696972131729, "rewards/total_composite/mean": 0.6291760802268982, "rewards/total_composite/std": 0.12751486897468567, "reward": 0.6291760802268982, "reward_std": 0.12751485407352448, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1923200637102127, "sampling/sampling_logp_difference/max": 1.5239753723144531, "sampling/importance_sampling_ratio/min": 0.217844158411026, "sampling/importance_sampling_ratio/mean": 1.0168901681900024, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4269536957144737, "clip_ratio/low_mean": 0.15119895059615374, "clip_ratio/low_min": 0.15119895059615374, "clip_ratio/high_mean": 0.012500000186264515, "clip_ratio/high_max": 0.012500000186264515, "clip_ratio/region_mean": 0.16369895078241825, "reward_total_mean": 0.6291760802268982, "reward_meter_mean": 0.8659785389900208, "reward_meter_std": 0.17113295197486877, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9889838099479675, "reward_repeat_soft_std": 0.008318226784467697, "reward_judge_quality_mean": 0.4937500059604645, "reward_judge_quality_std": 0.1728696972131729, "reward_total_composite_mean": 0.6291760802268982, "reward_total_composite_std": 0.12751486897468567} {"timestamp_utc": "2026-04-13T08:45:17Z", "mode": "train", "global_step": 553, "epoch": 0.05554997488699146, "loss": -0.0032, "grad_norm": 19.589412689208984, "learning_rate": 8.327272727272728e-06, "num_tokens": 986289.0, "completions/mean_length": 33.375, "completions/min_length": 30.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.375, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.435761958360672, "rewards/meter/std": 0.35650891065597534, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9923810958862305, "rewards/repeat_soft/std": 0.008589569479227066, "rewards/judge_quality/mean": 0.7987500429153442, "rewards/judge_quality/std": 0.22465451061725616, "rewards/total_composite/mean": 0.5534482002258301, "rewards/total_composite/std": 0.17337222397327423, "reward": 0.5534482002258301, "reward_std": 0.17337222397327423, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1868068426847458, "sampling/sampling_logp_difference/max": 2.0306060314178467, "sampling/importance_sampling_ratio/min": 0.13125595450401306, "sampling/importance_sampling_ratio/mean": 1.0006626844406128, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8828173168003559, "clip_ratio/low_mean": 0.0615382962860167, "clip_ratio/low_min": 0.0615382962860167, "clip_ratio/high_mean": 0.10045306198298931, "clip_ratio/high_max": 0.10045306198298931, "clip_ratio/region_mean": 0.161991358269006, "reward_total_mean": 0.5534482002258301, "reward_meter_mean": 0.435761958360672, "reward_meter_std": 0.35650891065597534, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9923810958862305, "reward_repeat_soft_std": 0.008589569479227066, "reward_judge_quality_mean": 0.7987500429153442, "reward_judge_quality_std": 0.22465451061725616, "reward_total_composite_mean": 0.5534482002258301, "reward_total_composite_std": 0.17337222397327423} {"timestamp_utc": "2026-04-13T08:45:28Z", "mode": "train", "global_step": 554, "epoch": 0.055650426921145156, "loss": -0.1029, "grad_norm": 3.118800163269043, "learning_rate": 8.324242424242425e-06, "num_tokens": 987937.0, "completions/mean_length": 101.0, "completions/min_length": 32.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 42.28571701049805, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.777889609336853, "rewards/meter/std": 0.38723862171173096, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9913482069969177, "rewards/repeat_soft/std": 0.013020490296185017, "rewards/judge_quality/mean": 0.4112499952316284, "rewards/judge_quality/std": 0.1797965168952942, "rewards/total_composite/mean": 0.5406044125556946, "rewards/total_composite/std": 0.2407016158103943, "reward": 0.5406044125556946, "reward_std": 0.2407016158103943, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21402639150619507, "sampling/sampling_logp_difference/max": 1.3933115005493164, "sampling/importance_sampling_ratio/min": 0.24825185537338257, "sampling/importance_sampling_ratio/mean": 1.00910222530365, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4998129308223724, "clip_ratio/low_mean": 0.021276595070958138, "clip_ratio/low_min": 0.021276595070958138, "clip_ratio/high_mean": 0.15984027367085218, "clip_ratio/high_max": 0.15984027367085218, "clip_ratio/region_mean": 0.18111686874181032, "reward_total_mean": 0.5406044125556946, "reward_meter_mean": 0.777889609336853, "reward_meter_std": 0.38723862171173096, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9913482069969177, "reward_repeat_soft_std": 0.013020490296185017, "reward_judge_quality_mean": 0.4112499952316284, "reward_judge_quality_std": 0.1797965168952942, "reward_total_composite_mean": 0.5406044125556946, "reward_total_composite_std": 0.2407016158103943} {"timestamp_utc": "2026-04-13T08:45:35Z", "mode": "train", "global_step": 555, "epoch": 0.055750878955298844, "loss": 0.0479, "grad_norm": 13.309117317199707, "learning_rate": 8.321212121212123e-06, "num_tokens": 989836.0, "completions/mean_length": 64.375, "completions/min_length": 44.0, "completions/max_length": 89.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.375, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 89.0, "rewards/meter/mean": 0.6408752202987671, "rewards/meter/std": 0.3405615985393524, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9968665242195129, "rewards/repeat_soft/std": 0.002312359632924199, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.6169316172599792, "rewards/total_composite/std": 0.16860747337341309, "reward": 0.6169316172599792, "reward_std": 0.16860747337341309, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19001272320747375, "sampling/sampling_logp_difference/max": 1.2530326843261719, "sampling/importance_sampling_ratio/min": 0.3489038050174713, "sampling/importance_sampling_ratio/mean": 1.0523494482040405, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7197429314255714, "clip_ratio/low_mean": 0.11605823412537575, "clip_ratio/low_min": 0.11605823412537575, "clip_ratio/high_mean": 0.07755850814282894, "clip_ratio/high_max": 0.07755850814282894, "clip_ratio/region_mean": 0.1936167422682047, "reward_total_mean": 0.6169316172599792, "reward_meter_mean": 0.6408752202987671, "reward_meter_std": 0.3405615985393524, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9968665242195129, "reward_repeat_soft_std": 0.002312359632924199, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.6169316172599792, "reward_total_composite_std": 0.16860747337341309} {"timestamp_utc": "2026-04-13T08:45:41Z", "mode": "train", "global_step": 556, "epoch": 0.05585133098945254, "loss": 0.1001, "grad_norm": 18.60330581665039, "learning_rate": 8.318181818181818e-06, "num_tokens": 991329.0, "completions/mean_length": 33.625, "completions/min_length": 28.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.625, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.9354280233383179, "rewards/meter/std": 0.11490907520055771, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9618691205978394, "rewards/repeat_soft/std": 0.03599108010530472, "rewards/judge_quality/mean": 0.5687500238418579, "rewards/judge_quality/std": 0.25176167488098145, "rewards/total_composite/mean": 0.681156575679779, "rewards/total_composite/std": 0.13293671607971191, "reward": 0.681156575679779, "reward_std": 0.13293671607971191, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17382174730300903, "sampling/sampling_logp_difference/max": 1.3681821823120117, "sampling/importance_sampling_ratio/min": 0.2602751553058624, "sampling/importance_sampling_ratio/mean": 1.0161211490631104, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2629192024469376, "clip_ratio/low_mean": 0.08075824566185474, "clip_ratio/low_min": 0.08075824566185474, "clip_ratio/high_mean": 0.05999729596078396, "clip_ratio/high_max": 0.05999729596078396, "clip_ratio/region_mean": 0.1407555416226387, "reward_total_mean": 0.681156575679779, "reward_meter_mean": 0.9354280233383179, "reward_meter_std": 0.11490907520055771, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9618691205978394, "reward_repeat_soft_std": 0.03599108010530472, "reward_judge_quality_mean": 0.5687500238418579, "reward_judge_quality_std": 0.25176167488098145, "reward_total_composite_mean": 0.681156575679779, "reward_total_composite_std": 0.13293671607971191} {"timestamp_utc": "2026-04-13T08:45:47Z", "mode": "train", "global_step": 557, "epoch": 0.055951783023606226, "loss": 0.0789, "grad_norm": 22.79527473449707, "learning_rate": 8.315151515151516e-06, "num_tokens": 992845.0, "completions/mean_length": 28.5, "completions/min_length": 27.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.5, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.45790940523147583, "rewards/meter/std": 0.3044460117816925, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.986666202545166, "rewards/repeat_soft/std": 0.014813925139605999, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.5273362398147583, "rewards/total_composite/std": 0.10774970799684525, "reward": 0.5273362398147583, "reward_std": 0.10774971544742584, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1532248854637146, "sampling/sampling_logp_difference/max": 1.1305646896362305, "sampling/importance_sampling_ratio/min": 0.32285091280937195, "sampling/importance_sampling_ratio/mean": 1.024466872215271, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8794346153736115, "clip_ratio/low_mean": 0.07793672196567059, "clip_ratio/low_min": 0.07793672196567059, "clip_ratio/high_mean": 0.12040343880653381, "clip_ratio/high_max": 0.12040343880653381, "clip_ratio/region_mean": 0.1983401607722044, "reward_total_mean": 0.5273362398147583, "reward_meter_mean": 0.45790940523147583, "reward_meter_std": 0.3044460117816925, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.986666202545166, "reward_repeat_soft_std": 0.014813925139605999, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.5273362398147583, "reward_total_composite_std": 0.10774970799684525} {"timestamp_utc": "2026-04-13T08:45:54Z", "mode": "train", "global_step": 558, "epoch": 0.05605223505775992, "loss": 0.0572, "grad_norm": 18.1457462310791, "learning_rate": 8.312121212121213e-06, "num_tokens": 994267.0, "completions/mean_length": 26.75, "completions/min_length": 20.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.75, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.7564411163330078, "rewards/meter/std": 0.42362114787101746, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9554247856140137, "rewards/repeat_soft/std": 0.017965074628591537, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.5618064403533936, "rewards/total_composite/std": 0.12254803627729416, "reward": 0.5618064403533936, "reward_std": 0.12254803627729416, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17637909948825836, "sampling/sampling_logp_difference/max": 1.416147232055664, "sampling/importance_sampling_ratio/min": 0.24264706671237946, "sampling/importance_sampling_ratio/mean": 1.0523557662963867, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3914649784564972, "clip_ratio/low_mean": 0.04214015230536461, "clip_ratio/low_min": 0.04214015230536461, "clip_ratio/high_mean": 0.1335457293316722, "clip_ratio/high_max": 0.1335457293316722, "clip_ratio/region_mean": 0.1756858816370368, "reward_total_mean": 0.5618064403533936, "reward_meter_mean": 0.7564411163330078, "reward_meter_std": 0.42362114787101746, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9554247856140137, "reward_repeat_soft_std": 0.017965074628591537, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.5618064403533936, "reward_total_composite_std": 0.12254803627729416} {"timestamp_utc": "2026-04-13T08:46:01Z", "mode": "train", "global_step": 559, "epoch": 0.05615268709191361, "loss": -0.1657, "grad_norm": 23.551433563232422, "learning_rate": 8.30909090909091e-06, "num_tokens": 995834.0, "completions/mean_length": 38.875, "completions/min_length": 9.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.875, "completions/min_terminated_length": 9.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.6795017719268799, "rewards/meter/std": 0.3432086408138275, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9923096299171448, "rewards/repeat_soft/std": 0.006110474467277527, "rewards/judge_quality/mean": 0.5325000286102295, "rewards/judge_quality/std": 0.26042547821998596, "rewards/total_composite/mean": 0.4991290867328644, "rewards/total_composite/std": 0.3228200376033783, "reward": 0.4991290867328644, "reward_std": 0.3228200376033783, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1697164624929428, "sampling/sampling_logp_difference/max": 1.7782261371612549, "sampling/importance_sampling_ratio/min": 0.1689375638961792, "sampling/importance_sampling_ratio/mean": 1.0004527568817139, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.101341262459755, "clip_ratio/low_mean": 0.0486111119389534, "clip_ratio/low_min": 0.0486111119389534, "clip_ratio/high_mean": 0.13180590840056539, "clip_ratio/high_max": 0.13180590840056539, "clip_ratio/region_mean": 0.18041702033951879, "reward_total_mean": 0.4991290867328644, "reward_meter_mean": 0.6795017719268799, "reward_meter_std": 0.3432086408138275, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9923096299171448, "reward_repeat_soft_std": 0.006110474467277527, "reward_judge_quality_mean": 0.5325000286102295, "reward_judge_quality_std": 0.26042547821998596, "reward_total_composite_mean": 0.4991290867328644, "reward_total_composite_std": 0.3228200376033783} {"timestamp_utc": "2026-04-13T08:46:07Z", "mode": "train", "global_step": 560, "epoch": 0.0562531391260673, "loss": 0.0135, "grad_norm": 22.024646759033203, "learning_rate": 8.306060606060606e-06, "num_tokens": 997180.0, "completions/mean_length": 19.25, "completions/min_length": 15.0, "completions/max_length": 22.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 19.25, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 22.0, "rewards/meter/mean": 0.8734430074691772, "rewards/meter/std": 0.31486058235168457, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9593343734741211, "rewards/repeat_soft/std": 0.008953613229095936, "rewards/judge_quality/mean": 0.5487500429153442, "rewards/judge_quality/std": 0.22937415540218353, "rewards/total_composite/mean": 0.6650569438934326, "rewards/total_composite/std": 0.1872844099998474, "reward": 0.6650569438934326, "reward_std": 0.1872844099998474, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16088703274726868, "sampling/sampling_logp_difference/max": 0.7771531343460083, "sampling/importance_sampling_ratio/min": 0.4651177227497101, "sampling/importance_sampling_ratio/mean": 1.0268609523773193, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4013171046972275, "clip_ratio/low_mean": 0.10809837467968464, "clip_ratio/low_min": 0.10809837467968464, "clip_ratio/high_mean": 0.03910427913069725, "clip_ratio/high_max": 0.03910427913069725, "clip_ratio/region_mean": 0.1472026538103819, "reward_total_mean": 0.6650569438934326, "reward_meter_mean": 0.8734430074691772, "reward_meter_std": 0.31486058235168457, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9593343734741211, "reward_repeat_soft_std": 0.008953613229095936, "reward_judge_quality_mean": 0.5487500429153442, "reward_judge_quality_std": 0.22937415540218353, "reward_total_composite_mean": 0.6650569438934326, "reward_total_composite_std": 0.1872844099998474} {"timestamp_utc": "2026-04-13T08:46:18Z", "mode": "train", "global_step": 561, "epoch": 0.056353591160221, "loss": -0.1958, "grad_norm": 3.2702996730804443, "learning_rate": 8.303030303030305e-06, "num_tokens": 999803.0, "completions/mean_length": 183.875, "completions/min_length": 101.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 137.0, "completions/min_terminated_length": 101.0, "completions/max_terminated_length": 183.0, "rewards/meter/mean": 0.8589916229248047, "rewards/meter/std": 0.22314368188381195, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.1035098284482956, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9383134841918945, "rewards/repeat_soft/std": 0.03509976714849472, "rewards/judge_quality/mean": 0.32499998807907104, "rewards/judge_quality/std": 0.15212775766849518, "rewards/total_composite/mean": 0.45598551630973816, "rewards/total_composite/std": 0.2027304619550705, "reward": 0.45598551630973816, "reward_std": 0.2027304619550705, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19196419417858124, "sampling/sampling_logp_difference/max": 1.3713769912719727, "sampling/importance_sampling_ratio/min": 0.2537572979927063, "sampling/importance_sampling_ratio/mean": 1.0575730800628662, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8707079887390137, "clip_ratio/low_mean": 0.018103448674082756, "clip_ratio/low_min": 0.018103448674082756, "clip_ratio/high_mean": 0.12402686849236488, "clip_ratio/high_max": 0.12402686849236488, "clip_ratio/region_mean": 0.14213031716644764, "reward_total_mean": 0.45598551630973816, "reward_meter_mean": 0.8589916229248047, "reward_meter_std": 0.22314368188381195, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.1035098284482956, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9383134841918945, "reward_repeat_soft_std": 0.03509976714849472, "reward_judge_quality_mean": 0.32499998807907104, "reward_judge_quality_std": 0.15212775766849518, "reward_total_composite_mean": 0.45598551630973816, "reward_total_composite_std": 0.2027304619550705} {"timestamp_utc": "2026-04-13T08:46:25Z", "mode": "train", "global_step": 562, "epoch": 0.056454043194374685, "loss": 0.0926, "grad_norm": 15.954366683959961, "learning_rate": 8.3e-06, "num_tokens": 1001398.0, "completions/mean_length": 42.375, "completions/min_length": 37.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.375, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.8383921384811401, "rewards/meter/std": 0.14216502010822296, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9863839149475098, "rewards/repeat_soft/std": 0.02003294602036476, "rewards/judge_quality/mean": 0.7325000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.7530303001403809, "rewards/total_composite/std": 0.1667768359184265, "reward": 0.7530303001403809, "reward_std": 0.1667768657207489, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18183524906635284, "sampling/sampling_logp_difference/max": 1.2515754699707031, "sampling/importance_sampling_ratio/min": 0.28605377674102783, "sampling/importance_sampling_ratio/mean": 1.0057904720306396, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3422640785574913, "clip_ratio/low_mean": 0.04274665657430887, "clip_ratio/low_min": 0.04274665657430887, "clip_ratio/high_mean": 0.09359593316912651, "clip_ratio/high_max": 0.09359593316912651, "clip_ratio/region_mean": 0.13634258974343538, "reward_total_mean": 0.7530303001403809, "reward_meter_mean": 0.8383921384811401, "reward_meter_std": 0.14216502010822296, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9863839149475098, "reward_repeat_soft_std": 0.02003294602036476, "reward_judge_quality_mean": 0.7325000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.7530303001403809, "reward_total_composite_std": 0.1667768359184265} {"timestamp_utc": "2026-04-13T08:46:33Z", "mode": "train", "global_step": 563, "epoch": 0.05655449522852838, "loss": 0.093, "grad_norm": 13.501004219055176, "learning_rate": 8.296969696969697e-06, "num_tokens": 1003244.0, "completions/mean_length": 56.75, "completions/min_length": 49.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.75, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.8068620562553406, "rewards/meter/std": 0.31526026129722595, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9576163291931152, "rewards/repeat_soft/std": 0.03529730811715126, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.48893821239471436, "rewards/total_composite/std": 0.21414178609848022, "reward": 0.48893821239471436, "reward_std": 0.21414178609848022, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16479219496250153, "sampling/sampling_logp_difference/max": 1.5833408832550049, "sampling/importance_sampling_ratio/min": 0.2052880972623825, "sampling/importance_sampling_ratio/mean": 1.0237668752670288, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0530619770288467, "clip_ratio/low_mean": 0.06251307297497988, "clip_ratio/low_min": 0.06251307297497988, "clip_ratio/high_mean": 0.10389220993965864, "clip_ratio/high_max": 0.10389220993965864, "clip_ratio/region_mean": 0.16640528291463852, "reward_total_mean": 0.48893821239471436, "reward_meter_mean": 0.8068620562553406, "reward_meter_std": 0.31526026129722595, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9576163291931152, "reward_repeat_soft_std": 0.03529730811715126, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.48893821239471436, "reward_total_composite_std": 0.21414178609848022} {"timestamp_utc": "2026-04-13T08:46:39Z", "mode": "train", "global_step": 564, "epoch": 0.05665494726268207, "loss": 0.0644, "grad_norm": 20.114315032958984, "learning_rate": 8.293939393939395e-06, "num_tokens": 1004749.0, "completions/mean_length": 22.125, "completions/min_length": 19.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.125, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.9016860127449036, "rewards/meter/std": 0.14367784559726715, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9148127436637878, "rewards/repeat_soft/std": 0.06743112951517105, "rewards/judge_quality/mean": 0.6737500429153442, "rewards/judge_quality/std": 0.263435423374176, "rewards/total_composite/mean": 0.7272079586982727, "rewards/total_composite/std": 0.1561218947172165, "reward": 0.7272079586982727, "reward_std": 0.15612190961837769, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16825152933597565, "sampling/sampling_logp_difference/max": 1.327385663986206, "sampling/importance_sampling_ratio/min": 0.3427193760871887, "sampling/importance_sampling_ratio/mean": 1.029595971107483, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.017693430185318, "clip_ratio/low_mean": 0.08651043241843581, "clip_ratio/low_min": 0.08651043241843581, "clip_ratio/high_mean": 0.0468017878010869, "clip_ratio/high_max": 0.0468017878010869, "clip_ratio/region_mean": 0.13331222021952271, "reward_total_mean": 0.7272079586982727, "reward_meter_mean": 0.9016860127449036, "reward_meter_std": 0.14367784559726715, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9148127436637878, "reward_repeat_soft_std": 0.06743112951517105, "reward_judge_quality_mean": 0.6737500429153442, "reward_judge_quality_std": 0.263435423374176, "reward_total_composite_mean": 0.7272079586982727, "reward_total_composite_std": 0.1561218947172165} {"timestamp_utc": "2026-04-13T08:46:45Z", "mode": "train", "global_step": 565, "epoch": 0.05675539929683576, "loss": -0.0434, "grad_norm": 18.368539810180664, "learning_rate": 8.290909090909092e-06, "num_tokens": 1006144.0, "completions/mean_length": 19.375, "completions/min_length": 17.0, "completions/max_length": 23.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 19.375, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 23.0, "rewards/meter/mean": 0.8119484186172485, "rewards/meter/std": 0.3543795049190521, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9605007171630859, "rewards/repeat_soft/std": 0.004598783794790506, "rewards/judge_quality/mean": 0.4025000035762787, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.524174690246582, "rewards/total_composite/std": 0.2167847901582718, "reward": 0.524174690246582, "reward_std": 0.2167847752571106, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19802412390708923, "sampling/sampling_logp_difference/max": 1.334218978881836, "sampling/importance_sampling_ratio/min": 0.263363778591156, "sampling/importance_sampling_ratio/mean": 0.9816744923591614, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5238390415906906, "clip_ratio/low_mean": 0.03851540666073561, "clip_ratio/low_min": 0.03851540666073561, "clip_ratio/high_mean": 0.1667059948667884, "clip_ratio/high_max": 0.1667059948667884, "clip_ratio/region_mean": 0.205221401527524, "reward_total_mean": 0.524174690246582, "reward_meter_mean": 0.8119484186172485, "reward_meter_std": 0.3543795049190521, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9605007171630859, "reward_repeat_soft_std": 0.004598783794790506, "reward_judge_quality_mean": 0.4025000035762787, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.524174690246582, "reward_total_composite_std": 0.2167847901582718} {"timestamp_utc": "2026-04-13T08:46:51Z", "mode": "train", "global_step": 566, "epoch": 0.05685585133098945, "loss": 0.1211, "grad_norm": 20.236555099487305, "learning_rate": 8.287878787878787e-06, "num_tokens": 1007630.0, "completions/mean_length": 41.75, "completions/min_length": 33.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.75, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.8665573596954346, "rewards/meter/std": 0.3077116906642914, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9985834360122681, "rewards/repeat_soft/std": 0.003662696573883295, "rewards/judge_quality/mean": 0.5487499833106995, "rewards/judge_quality/std": 0.22937417030334473, "rewards/total_composite/mean": 0.6672229170799255, "rewards/total_composite/std": 0.18799370527267456, "reward": 0.6672229170799255, "reward_std": 0.18799370527267456, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19202497601509094, "sampling/sampling_logp_difference/max": 1.9503707885742188, "sampling/importance_sampling_ratio/min": 0.1422213315963745, "sampling/importance_sampling_ratio/mean": 1.0313491821289062, "sampling/importance_sampling_ratio/max": 1.9711854457855225, "entropy": 1.863529920578003, "clip_ratio/low_mean": 0.12671365309506655, "clip_ratio/low_min": 0.12671365309506655, "clip_ratio/high_mean": 0.038888889364898205, "clip_ratio/high_max": 0.038888889364898205, "clip_ratio/region_mean": 0.16560254245996475, "reward_total_mean": 0.6672229170799255, "reward_meter_mean": 0.8665573596954346, "reward_meter_std": 0.3077116906642914, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9985834360122681, "reward_repeat_soft_std": 0.003662696573883295, "reward_judge_quality_mean": 0.5487499833106995, "reward_judge_quality_std": 0.22937417030334473, "reward_total_composite_mean": 0.6672229170799255, "reward_total_composite_std": 0.18799370527267456} {"timestamp_utc": "2026-04-13T08:46:57Z", "mode": "train", "global_step": 567, "epoch": 0.056956303365143145, "loss": 0.0495, "grad_norm": 18.38676643371582, "learning_rate": 8.284848484848486e-06, "num_tokens": 1009144.0, "completions/mean_length": 38.25, "completions/min_length": 34.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.25, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.9538601636886597, "rewards/meter/std": 0.10505902767181396, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.980604887008667, "rewards/repeat_soft/std": 0.019677551463246346, "rewards/judge_quality/mean": 0.7487500309944153, "rewards/judge_quality/std": 0.21853001415729523, "rewards/total_composite/mean": 0.8193771243095398, "rewards/total_composite/std": 0.16228778660297394, "reward": 0.8193771243095398, "reward_std": 0.16228780150413513, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1814090460538864, "sampling/sampling_logp_difference/max": 1.8432121276855469, "sampling/importance_sampling_ratio/min": 0.15830810368061066, "sampling/importance_sampling_ratio/mean": 1.0160726308822632, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1473611667752266, "clip_ratio/low_mean": 0.06210801098495722, "clip_ratio/low_min": 0.06210801098495722, "clip_ratio/high_mean": 0.1015858231112361, "clip_ratio/high_max": 0.1015858231112361, "clip_ratio/region_mean": 0.1636938340961933, "reward_total_mean": 0.8193771243095398, "reward_meter_mean": 0.9538601636886597, "reward_meter_std": 0.10505902767181396, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.980604887008667, "reward_repeat_soft_std": 0.019677551463246346, "reward_judge_quality_mean": 0.7487500309944153, "reward_judge_quality_std": 0.21853001415729523, "reward_total_composite_mean": 0.8193771243095398, "reward_total_composite_std": 0.16228778660297394} {"timestamp_utc": "2026-04-13T08:47:04Z", "mode": "train", "global_step": 568, "epoch": 0.05705675539929683, "loss": 0.0749, "grad_norm": 15.373639106750488, "learning_rate": 8.281818181818182e-06, "num_tokens": 1010964.0, "completions/mean_length": 48.5, "completions/min_length": 42.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.5, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9488092660903931, "rewards/meter/std": 0.10913003236055374, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.937246561050415, "rewards/repeat_soft/std": 0.07722450792789459, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.585860550403595, "rewards/total_composite/std": 0.053448330610990524, "reward": 0.585860550403595, "reward_std": 0.05344833433628082, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20093071460723877, "sampling/sampling_logp_difference/max": 1.6151256561279297, "sampling/importance_sampling_ratio/min": 0.19886568188667297, "sampling/importance_sampling_ratio/mean": 0.9998076558113098, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3005739152431488, "clip_ratio/low_mean": 0.04811224527657032, "clip_ratio/low_min": 0.04811224527657032, "clip_ratio/high_mean": 0.1577774751931429, "clip_ratio/high_max": 0.1577774751931429, "clip_ratio/region_mean": 0.2058897204697132, "reward_total_mean": 0.585860550403595, "reward_meter_mean": 0.9488092660903931, "reward_meter_std": 0.10913003236055374, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.937246561050415, "reward_repeat_soft_std": 0.07722450792789459, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.585860550403595, "reward_total_composite_std": 0.053448330610990524} {"timestamp_utc": "2026-04-13T08:47:10Z", "mode": "train", "global_step": 569, "epoch": 0.05715720743345053, "loss": 0.0922, "grad_norm": 23.371784210205078, "learning_rate": 8.27878787878788e-06, "num_tokens": 1012403.0, "completions/mean_length": 29.875, "completions/min_length": 23.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.875, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.6001070141792297, "rewards/meter/std": 0.4020649492740631, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9339101314544678, "rewards/repeat_soft/std": 0.06879226118326187, "rewards/judge_quality/mean": 0.8650000095367432, "rewards/judge_quality/std": 0.16801361739635468, "rewards/total_composite/mean": 0.6974748969078064, "rewards/total_composite/std": 0.24909807741641998, "reward": 0.6974748969078064, "reward_std": 0.2490980625152588, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15942718088626862, "sampling/sampling_logp_difference/max": 1.461019515991211, "sampling/importance_sampling_ratio/min": 0.23199963569641113, "sampling/importance_sampling_ratio/mean": 0.9991790652275085, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7846294343471527, "clip_ratio/low_mean": 0.039762435015290976, "clip_ratio/low_min": 0.039762435015290976, "clip_ratio/high_mean": 0.06495859334245324, "clip_ratio/high_max": 0.06495859334245324, "clip_ratio/region_mean": 0.10472102835774422, "reward_total_mean": 0.6974748969078064, "reward_meter_mean": 0.6001070141792297, "reward_meter_std": 0.4020649492740631, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9339101314544678, "reward_repeat_soft_std": 0.06879226118326187, "reward_judge_quality_mean": 0.8650000095367432, "reward_judge_quality_std": 0.16801361739635468, "reward_total_composite_mean": 0.6974748969078064, "reward_total_composite_std": 0.24909807741641998} {"timestamp_utc": "2026-04-13T08:47:16Z", "mode": "train", "global_step": 570, "epoch": 0.05725765946760422, "loss": 0.0473, "grad_norm": 17.427688598632812, "learning_rate": 8.275757575757577e-06, "num_tokens": 1013981.0, "completions/mean_length": 40.25, "completions/min_length": 30.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.25, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.5616979598999023, "rewards/meter/std": 0.3486446738243103, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9733009338378906, "rewards/repeat_soft/std": 0.028935855254530907, "rewards/judge_quality/mean": 0.543749988079071, "rewards/judge_quality/std": 0.14647649228572845, "rewards/total_composite/mean": 0.48278260231018066, "rewards/total_composite/std": 0.2439933717250824, "reward": 0.48278260231018066, "reward_std": 0.2439933717250824, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16998045146465302, "sampling/sampling_logp_difference/max": 1.2940139770507812, "sampling/importance_sampling_ratio/min": 0.2741680443286896, "sampling/importance_sampling_ratio/mean": 1.0163642168045044, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3805529028177261, "clip_ratio/low_mean": 0.06506410427391529, "clip_ratio/low_min": 0.06506410427391529, "clip_ratio/high_mean": 0.10038667172193527, "clip_ratio/high_max": 0.10038667172193527, "clip_ratio/region_mean": 0.16545077599585056, "reward_total_mean": 0.48278260231018066, "reward_meter_mean": 0.5616979598999023, "reward_meter_std": 0.3486446738243103, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9733009338378906, "reward_repeat_soft_std": 0.028935855254530907, "reward_judge_quality_mean": 0.543749988079071, "reward_judge_quality_std": 0.14647649228572845, "reward_total_composite_mean": 0.48278260231018066, "reward_total_composite_std": 0.2439933717250824} {"timestamp_utc": "2026-04-13T08:47:27Z", "mode": "train", "global_step": 571, "epoch": 0.05735811150175791, "loss": -0.1987, "grad_norm": 3.877699613571167, "learning_rate": 8.272727272727274e-06, "num_tokens": 1016407.0, "completions/mean_length": 154.25, "completions/min_length": 84.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 103.14286041259766, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 135.0, "rewards/meter/mean": 0.49068641662597656, "rewards/meter/std": 0.2176356166601181, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.993622899055481, "rewards/repeat_soft/std": 0.003882135497406125, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.2121320515871048, "rewards/total_composite/mean": 0.4497496485710144, "rewards/total_composite/std": 0.20662787556648254, "reward": 0.4497496485710144, "reward_std": 0.20662787556648254, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19983239471912384, "sampling/sampling_logp_difference/max": 2.1510086059570312, "sampling/importance_sampling_ratio/min": 0.11636672168970108, "sampling/importance_sampling_ratio/mean": 1.033197045326233, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.37043796479702, "clip_ratio/low_mean": 0.06433553248643875, "clip_ratio/low_min": 0.06433553248643875, "clip_ratio/high_mean": 0.08608468901365995, "clip_ratio/high_max": 0.08608468901365995, "clip_ratio/region_mean": 0.1504202215000987, "reward_total_mean": 0.4497496485710144, "reward_meter_mean": 0.49068641662597656, "reward_meter_std": 0.2176356166601181, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.993622899055481, "reward_repeat_soft_std": 0.003882135497406125, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.2121320515871048, "reward_total_composite_mean": 0.4497496485710144, "reward_total_composite_std": 0.20662787556648254} {"timestamp_utc": "2026-04-13T08:47:33Z", "mode": "train", "global_step": 572, "epoch": 0.057458563535911604, "loss": 0.1492, "grad_norm": 16.63024139404297, "learning_rate": 8.269696969696971e-06, "num_tokens": 1018101.0, "completions/mean_length": 59.75, "completions/min_length": 46.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.75, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.7083966135978699, "rewards/meter/std": 0.357369601726532, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9095284342765808, "rewards/repeat_soft/std": 0.0766533613204956, "rewards/judge_quality/mean": 0.5325000286102295, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.5992577075958252, "rewards/total_composite/std": 0.1747995764017105, "reward": 0.5992577075958252, "reward_std": 0.17479956150054932, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16766788065433502, "sampling/sampling_logp_difference/max": 1.4486650228500366, "sampling/importance_sampling_ratio/min": 0.23488366603851318, "sampling/importance_sampling_ratio/mean": 1.0143681764602661, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0160523355007172, "clip_ratio/low_mean": 0.06509399600327015, "clip_ratio/low_min": 0.06509399600327015, "clip_ratio/high_mean": 0.09081595111638308, "clip_ratio/high_max": 0.09081595111638308, "clip_ratio/region_mean": 0.15590994711965322, "reward_total_mean": 0.5992577075958252, "reward_meter_mean": 0.7083966135978699, "reward_meter_std": 0.357369601726532, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9095284342765808, "reward_repeat_soft_std": 0.0766533613204956, "reward_judge_quality_mean": 0.5325000286102295, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.5992577075958252, "reward_total_composite_std": 0.1747995764017105} {"timestamp_utc": "2026-04-13T08:47:45Z", "mode": "train", "global_step": 573, "epoch": 0.05755901557006529, "loss": -0.0601, "grad_norm": 5.129828929901123, "learning_rate": 8.266666666666667e-06, "num_tokens": 1019906.0, "completions/mean_length": 110.625, "completions/min_length": 45.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 53.28571701049805, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.8684149980545044, "rewards/meter/std": 0.3483780026435852, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.7412621974945068, "rewards/repeat_soft/std": 0.20249877870082855, "rewards/judge_quality/mean": 0.33125001192092896, "rewards/judge_quality/std": 0.13715866208076477, "rewards/total_composite/mean": 0.3450530469417572, "rewards/total_composite/std": 0.2909436523914337, "reward": 0.3450530469417572, "reward_std": 0.29094362258911133, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1584600806236267, "sampling/sampling_logp_difference/max": 1.4114131927490234, "sampling/importance_sampling_ratio/min": 0.24379850924015045, "sampling/importance_sampling_ratio/mean": 1.0394682884216309, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.247895561158657, "clip_ratio/low_mean": 0.054653484374284744, "clip_ratio/low_min": 0.054653484374284744, "clip_ratio/high_mean": 0.07300596591085196, "clip_ratio/high_max": 0.07300596591085196, "clip_ratio/region_mean": 0.1276594502851367, "reward_total_mean": 0.3450530469417572, "reward_meter_mean": 0.8684149980545044, "reward_meter_std": 0.3483780026435852, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.7412621974945068, "reward_repeat_soft_std": 0.20249877870082855, "reward_judge_quality_mean": 0.33125001192092896, "reward_judge_quality_std": 0.13715866208076477, "reward_total_composite_mean": 0.3450530469417572, "reward_total_composite_std": 0.2909436523914337} {"timestamp_utc": "2026-04-13T08:47:51Z", "mode": "train", "global_step": 574, "epoch": 0.057659467604218986, "loss": 0.0174, "grad_norm": 33.07956314086914, "learning_rate": 8.263636363636366e-06, "num_tokens": 1021418.0, "completions/mean_length": 41.0, "completions/min_length": 30.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.0, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.6174425482749939, "rewards/meter/std": 0.37483736872673035, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9980881810188293, "rewards/repeat_soft/std": 0.005338655319064856, "rewards/judge_quality/mean": 0.5137500166893005, "rewards/judge_quality/std": 0.1277204155921936, "rewards/total_composite/mean": 0.5354558229446411, "rewards/total_composite/std": 0.09790372103452682, "reward": 0.5354558229446411, "reward_std": 0.09790373593568802, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17352846264839172, "sampling/sampling_logp_difference/max": 1.3274812698364258, "sampling/importance_sampling_ratio/min": 0.2651442587375641, "sampling/importance_sampling_ratio/mean": 0.9985684156417847, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8166610077023506, "clip_ratio/low_mean": 0.08509136270731688, "clip_ratio/low_min": 0.08509136270731688, "clip_ratio/high_mean": 0.08918650913983583, "clip_ratio/high_max": 0.08918650913983583, "clip_ratio/region_mean": 0.1742778718471527, "reward_total_mean": 0.5354558229446411, "reward_meter_mean": 0.6174425482749939, "reward_meter_std": 0.37483736872673035, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9980881810188293, "reward_repeat_soft_std": 0.005338655319064856, "reward_judge_quality_mean": 0.5137500166893005, "reward_judge_quality_std": 0.1277204155921936, "reward_total_composite_mean": 0.5354558229446411, "reward_total_composite_std": 0.09790372103452682} {"timestamp_utc": "2026-04-13T08:47:57Z", "mode": "train", "global_step": 575, "epoch": 0.057759919638372674, "loss": 0.0678, "grad_norm": 20.26004409790039, "learning_rate": 8.260606060606061e-06, "num_tokens": 1023041.0, "completions/mean_length": 32.875, "completions/min_length": 25.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.875, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.830098032951355, "rewards/meter/std": 0.3257444500923157, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9855596423149109, "rewards/repeat_soft/std": 0.018953589722514153, "rewards/judge_quality/mean": 0.5900000333786011, "rewards/judge_quality/std": 0.27994900941848755, "rewards/total_composite/mean": 0.6754432320594788, "rewards/total_composite/std": 0.20469191670417786, "reward": 0.6754432320594788, "reward_std": 0.20469191670417786, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16457249224185944, "sampling/sampling_logp_difference/max": 1.3743727207183838, "sampling/importance_sampling_ratio/min": 0.2529982328414917, "sampling/importance_sampling_ratio/mean": 0.9803752303123474, "sampling/importance_sampling_ratio/max": 1.7883014678955078, "entropy": 1.0091527700424194, "clip_ratio/low_mean": 0.03716179495677352, "clip_ratio/low_min": 0.03716179495677352, "clip_ratio/high_mean": 0.036587481619790196, "clip_ratio/high_max": 0.036587481619790196, "clip_ratio/region_mean": 0.07374927657656372, "reward_total_mean": 0.6754432320594788, "reward_meter_mean": 0.830098032951355, "reward_meter_std": 0.3257444500923157, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9855596423149109, "reward_repeat_soft_std": 0.018953589722514153, "reward_judge_quality_mean": 0.5900000333786011, "reward_judge_quality_std": 0.27994900941848755, "reward_total_composite_mean": 0.6754432320594788, "reward_total_composite_std": 0.20469191670417786} {"timestamp_utc": "2026-04-13T08:48:03Z", "mode": "train", "global_step": 576, "epoch": 0.05786037167252637, "loss": 0.0122, "grad_norm": 29.855863571166992, "learning_rate": 8.257575757575758e-06, "num_tokens": 1024508.0, "completions/mean_length": 32.375, "completions/min_length": 29.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.375, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.8965328931808472, "rewards/meter/std": 0.14603206515312195, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9723629951477051, "rewards/repeat_soft/std": 0.042240433394908905, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5822746157646179, "rewards/total_composite/std": 0.04208853840827942, "reward": 0.5822746157646179, "reward_std": 0.042088523507118225, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17073535919189453, "sampling/sampling_logp_difference/max": 2.2079977989196777, "sampling/importance_sampling_ratio/min": 0.10992051661014557, "sampling/importance_sampling_ratio/mean": 1.0111312866210938, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9421104788780212, "clip_ratio/low_mean": 0.03553427383303642, "clip_ratio/low_min": 0.03553427383303642, "clip_ratio/high_mean": 0.10378700867295265, "clip_ratio/high_max": 0.10378700867295265, "clip_ratio/region_mean": 0.13932128250598907, "reward_total_mean": 0.5822746157646179, "reward_meter_mean": 0.8965328931808472, "reward_meter_std": 0.14603206515312195, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9723629951477051, "reward_repeat_soft_std": 0.042240433394908905, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5822746157646179, "reward_total_composite_std": 0.04208853840827942} {"timestamp_utc": "2026-04-13T08:48:14Z", "mode": "train", "global_step": 577, "epoch": 0.05796082370668006, "loss": -0.1619, "grad_norm": 4.448294162750244, "learning_rate": 8.254545454545456e-06, "num_tokens": 1026542.0, "completions/mean_length": 127.25, "completions/min_length": 60.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 72.28572082519531, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 89.0, "rewards/meter/mean": 0.7502434253692627, "rewards/meter/std": 0.2678214907646179, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9235777854919434, "rewards/repeat_soft/std": 0.09846273064613342, "rewards/judge_quality/mean": 0.44875001907348633, "rewards/judge_quality/std": 0.2105392962694168, "rewards/total_composite/mean": 0.5139748454093933, "rewards/total_composite/std": 0.2346297651529312, "reward": 0.5139748454093933, "reward_std": 0.2346297651529312, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20501717925071716, "sampling/sampling_logp_difference/max": 1.633583664894104, "sampling/importance_sampling_ratio/min": 0.1952286809682846, "sampling/importance_sampling_ratio/mean": 1.032249093055725, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.357079640030861, "clip_ratio/low_mean": 0.0470130518078804, "clip_ratio/low_min": 0.0470130518078804, "clip_ratio/high_mean": 0.11227748077362776, "clip_ratio/high_max": 0.11227748077362776, "clip_ratio/region_mean": 0.15929053258150816, "reward_total_mean": 0.5139748454093933, "reward_meter_mean": 0.7502434253692627, "reward_meter_std": 0.2678214907646179, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9235777854919434, "reward_repeat_soft_std": 0.09846273064613342, "reward_judge_quality_mean": 0.44875001907348633, "reward_judge_quality_std": 0.2105392962694168, "reward_total_composite_mean": 0.5139748454093933, "reward_total_composite_std": 0.2346297651529312} {"timestamp_utc": "2026-04-13T08:48:20Z", "mode": "train", "global_step": 578, "epoch": 0.05806127574083375, "loss": 0.0497, "grad_norm": 16.69968032836914, "learning_rate": 8.251515151515153e-06, "num_tokens": 1028078.0, "completions/mean_length": 35.0, "completions/min_length": 31.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.0, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.9882268905639648, "rewards/meter/std": 0.004939976613968611, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9579628705978394, "rewards/repeat_soft/std": 0.053920794278383255, "rewards/judge_quality/mean": 0.5974999666213989, "rewards/judge_quality/std": 0.220891073346138, "rewards/total_composite/mean": 0.7276318073272705, "rewards/total_composite/std": 0.14510180056095123, "reward": 0.7276318073272705, "reward_std": 0.14510181546211243, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2057153284549713, "sampling/sampling_logp_difference/max": 1.1017704010009766, "sampling/importance_sampling_ratio/min": 0.33228227496147156, "sampling/importance_sampling_ratio/mean": 1.0183676481246948, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7146768867969513, "clip_ratio/low_mean": 0.08770505990833044, "clip_ratio/low_min": 0.08770505990833044, "clip_ratio/high_mean": 0.07052139192819595, "clip_ratio/high_max": 0.07052139192819595, "clip_ratio/region_mean": 0.1582264518365264, "reward_total_mean": 0.7276318073272705, "reward_meter_mean": 0.9882268905639648, "reward_meter_std": 0.004939976613968611, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9579628705978394, "reward_repeat_soft_std": 0.053920794278383255, "reward_judge_quality_mean": 0.5974999666213989, "reward_judge_quality_std": 0.220891073346138, "reward_total_composite_mean": 0.7276318073272705, "reward_total_composite_std": 0.14510180056095123} {"timestamp_utc": "2026-04-13T08:48:27Z", "mode": "train", "global_step": 579, "epoch": 0.058161727774987446, "loss": -0.0024, "grad_norm": 20.66975212097168, "learning_rate": 8.248484848484848e-06, "num_tokens": 1029624.0, "completions/mean_length": 37.25, "completions/min_length": 33.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.25, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.26896071434020996, "rewards/meter/std": 0.26846253871917725, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9975525736808777, "rewards/repeat_soft/std": 0.0028212887700647116, "rewards/judge_quality/mean": 0.6012499928474426, "rewards/judge_quality/std": 0.22350697219371796, "rewards/total_composite/mean": 0.4431477189064026, "rewards/total_composite/std": 0.08015746623277664, "reward": 0.4431477189064026, "reward_std": 0.08015745133161545, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17363011837005615, "sampling/sampling_logp_difference/max": 1.3714125156402588, "sampling/importance_sampling_ratio/min": 0.2537482678890228, "sampling/importance_sampling_ratio/mean": 1.0310235023498535, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.264290452003479, "clip_ratio/low_mean": 0.08566673472523689, "clip_ratio/low_min": 0.08566673472523689, "clip_ratio/high_mean": 0.052020203322172165, "clip_ratio/high_max": 0.052020203322172165, "clip_ratio/region_mean": 0.13768693804740906, "reward_total_mean": 0.4431477189064026, "reward_meter_mean": 0.26896071434020996, "reward_meter_std": 0.26846253871917725, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9975525736808777, "reward_repeat_soft_std": 0.0028212887700647116, "reward_judge_quality_mean": 0.6012499928474426, "reward_judge_quality_std": 0.22350697219371796, "reward_total_composite_mean": 0.4431477189064026, "reward_total_composite_std": 0.08015746623277664} {"timestamp_utc": "2026-04-13T08:48:34Z", "mode": "train", "global_step": 580, "epoch": 0.05826217980914113, "loss": 0.0927, "grad_norm": 9.298893928527832, "learning_rate": 8.245454545454546e-06, "num_tokens": 1031667.0, "completions/mean_length": 76.375, "completions/min_length": 61.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.375, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.963345468044281, "rewards/meter/std": 0.05591030418872833, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6659365892410278, "rewards/repeat_soft/std": 0.18439270555973053, "rewards/judge_quality/mean": 0.3812499940395355, "rewards/judge_quality/std": 0.17266297340393066, "rewards/total_composite/mean": 0.5362281799316406, "rewards/total_composite/std": 0.12325633317232132, "reward": 0.5362281799316406, "reward_std": 0.12325632572174072, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11173097044229507, "sampling/sampling_logp_difference/max": 1.7862787246704102, "sampling/importance_sampling_ratio/min": 0.1675826460123062, "sampling/importance_sampling_ratio/mean": 1.0131728649139404, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8593889139592648, "clip_ratio/low_mean": 0.02751857414841652, "clip_ratio/low_min": 0.02751857414841652, "clip_ratio/high_mean": 0.07695895247161388, "clip_ratio/high_max": 0.07695895247161388, "clip_ratio/region_mean": 0.1044775266200304, "reward_total_mean": 0.5362281799316406, "reward_meter_mean": 0.963345468044281, "reward_meter_std": 0.05591030418872833, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6659365892410278, "reward_repeat_soft_std": 0.18439270555973053, "reward_judge_quality_mean": 0.3812499940395355, "reward_judge_quality_std": 0.17266297340393066, "reward_total_composite_mean": 0.5362281799316406, "reward_total_composite_std": 0.12325633317232132} {"timestamp_utc": "2026-04-13T08:48:40Z", "mode": "train", "global_step": 581, "epoch": 0.05836263184329483, "loss": -0.0275, "grad_norm": 17.0106143951416, "learning_rate": 8.242424242424243e-06, "num_tokens": 1033521.0, "completions/mean_length": 49.75, "completions/min_length": 38.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.75, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.838412344455719, "rewards/meter/std": 0.3349061608314514, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9712759256362915, "rewards/repeat_soft/std": 0.0544203482568264, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.57071852684021, "rewards/total_composite/std": 0.35998836159706116, "reward": 0.57071852684021, "reward_std": 0.35998836159706116, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1976938396692276, "sampling/sampling_logp_difference/max": 1.3331255912780762, "sampling/importance_sampling_ratio/min": 0.26365190744400024, "sampling/importance_sampling_ratio/mean": 1.0169309377670288, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2996100634336472, "clip_ratio/low_mean": 0.04820086061954498, "clip_ratio/low_min": 0.04820086061954498, "clip_ratio/high_mean": 0.1340278284624219, "clip_ratio/high_max": 0.1340278284624219, "clip_ratio/region_mean": 0.18222868908196688, "reward_total_mean": 0.57071852684021, "reward_meter_mean": 0.838412344455719, "reward_meter_std": 0.3349061608314514, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9712759256362915, "reward_repeat_soft_std": 0.0544203482568264, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.57071852684021, "reward_total_composite_std": 0.35998836159706116} {"timestamp_utc": "2026-04-13T08:48:46Z", "mode": "train", "global_step": 582, "epoch": 0.058463083877448516, "loss": 0.1068, "grad_norm": 25.380718231201172, "learning_rate": 8.23939393939394e-06, "num_tokens": 1035309.0, "completions/mean_length": 48.5, "completions/min_length": 46.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.5, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.7526752352714539, "rewards/meter/std": 0.32828620076179504, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8937437534332275, "rewards/repeat_soft/std": 0.08044855296611786, "rewards/judge_quality/mean": 0.5699999928474426, "rewards/judge_quality/std": 0.16035676002502441, "rewards/total_composite/mean": 0.611682653427124, "rewards/total_composite/std": 0.1384030431509018, "reward": 0.611682653427124, "reward_std": 0.1384030282497406, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18637044727802277, "sampling/sampling_logp_difference/max": 1.5626764297485352, "sampling/importance_sampling_ratio/min": 0.20957441627979279, "sampling/importance_sampling_ratio/mean": 1.0096079111099243, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1540735960006714, "clip_ratio/low_mean": 0.08840708434581757, "clip_ratio/low_min": 0.08840708434581757, "clip_ratio/high_mean": 0.06679530814290047, "clip_ratio/high_max": 0.06679530814290047, "clip_ratio/region_mean": 0.15520239248871803, "reward_total_mean": 0.611682653427124, "reward_meter_mean": 0.7526752352714539, "reward_meter_std": 0.32828620076179504, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8937437534332275, "reward_repeat_soft_std": 0.08044855296611786, "reward_judge_quality_mean": 0.5699999928474426, "reward_judge_quality_std": 0.16035676002502441, "reward_total_composite_mean": 0.611682653427124, "reward_total_composite_std": 0.1384030431509018} {"timestamp_utc": "2026-04-13T08:48:52Z", "mode": "train", "global_step": 583, "epoch": 0.05856353591160221, "loss": 0.016, "grad_norm": 20.738035202026367, "learning_rate": 8.236363636363637e-06, "num_tokens": 1036800.0, "completions/mean_length": 28.375, "completions/min_length": 23.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.375, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.8921211957931519, "rewards/meter/std": 0.17309178411960602, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.933234691619873, "rewards/repeat_soft/std": 0.04354261979460716, "rewards/judge_quality/mean": 0.5487499833106995, "rewards/judge_quality/std": 0.22937417030334473, "rewards/total_composite/mean": 0.6594759821891785, "rewards/total_composite/std": 0.14793506264686584, "reward": 0.6594759821891785, "reward_std": 0.14793504774570465, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20247289538383484, "sampling/sampling_logp_difference/max": 1.9466551542282104, "sampling/importance_sampling_ratio/min": 0.14275075495243073, "sampling/importance_sampling_ratio/mean": 0.9954608678817749, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0263854786753654, "clip_ratio/low_mean": 0.12533949594944715, "clip_ratio/low_min": 0.12533949594944715, "clip_ratio/high_mean": 0.03525641094893217, "clip_ratio/high_max": 0.03525641094893217, "clip_ratio/region_mean": 0.16059590689837933, "reward_total_mean": 0.6594759821891785, "reward_meter_mean": 0.8921211957931519, "reward_meter_std": 0.17309178411960602, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.933234691619873, "reward_repeat_soft_std": 0.04354261979460716, "reward_judge_quality_mean": 0.5487499833106995, "reward_judge_quality_std": 0.22937417030334473, "reward_total_composite_mean": 0.6594759821891785, "reward_total_composite_std": 0.14793506264686584} {"timestamp_utc": "2026-04-13T08:49:04Z", "mode": "train", "global_step": 584, "epoch": 0.058663987945755905, "loss": -0.1672, "grad_norm": 3.1960246562957764, "learning_rate": 8.233333333333335e-06, "num_tokens": 1038841.0, "completions/mean_length": 139.125, "completions/min_length": 80.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 85.85714721679688, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.7035621404647827, "rewards/meter/std": 0.4134138822555542, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.2828427255153656, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8031180500984192, "rewards/repeat_soft/std": 0.1281014382839203, "rewards/judge_quality/mean": 0.3687499761581421, "rewards/judge_quality/std": 0.1940866857767105, "rewards/total_composite/mean": 0.4458889067173004, "rewards/total_composite/std": 0.20599308609962463, "reward": 0.4458889067173004, "reward_std": 0.20599307119846344, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15093861520290375, "sampling/sampling_logp_difference/max": 2.0359926223754883, "sampling/importance_sampling_ratio/min": 0.1305508315563202, "sampling/importance_sampling_ratio/mean": 1.0371010303497314, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3707344383001328, "clip_ratio/low_mean": 0.04888310004025698, "clip_ratio/low_min": 0.04888310004025698, "clip_ratio/high_mean": 0.08071568608283997, "clip_ratio/high_max": 0.08071568608283997, "clip_ratio/region_mean": 0.12959878612309694, "reward_total_mean": 0.4458889067173004, "reward_meter_mean": 0.7035621404647827, "reward_meter_std": 0.4134138822555542, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.2828427255153656, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8031180500984192, "reward_repeat_soft_std": 0.1281014382839203, "reward_judge_quality_mean": 0.3687499761581421, "reward_judge_quality_std": 0.1940866857767105, "reward_total_composite_mean": 0.4458889067173004, "reward_total_composite_std": 0.20599308609962463} {"timestamp_utc": "2026-04-13T08:49:10Z", "mode": "train", "global_step": 585, "epoch": 0.05876443997990959, "loss": -0.0034, "grad_norm": 24.6851806640625, "learning_rate": 8.23030303030303e-06, "num_tokens": 1040186.0, "completions/mean_length": 17.125, "completions/min_length": 15.0, "completions/max_length": 20.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 17.125, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 20.0, "rewards/meter/mean": 0.4121133089065552, "rewards/meter/std": 0.42796626687049866, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.4591677784919739, "rewards/total_composite/std": 0.11875561624765396, "reward": 0.4591677784919739, "reward_std": 0.11875561624765396, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09549973160028458, "sampling/sampling_logp_difference/max": 0.8919789791107178, "sampling/importance_sampling_ratio/min": 0.40984389185905457, "sampling/importance_sampling_ratio/mean": 1.0326519012451172, "sampling/importance_sampling_ratio/max": 1.9693711996078491, "entropy": 0.662678062915802, "clip_ratio/low_mean": 0.014756944496184587, "clip_ratio/low_min": 0.014756944496184587, "clip_ratio/high_mean": 0.0350694446824491, "clip_ratio/high_max": 0.0350694446824491, "clip_ratio/region_mean": 0.04982638917863369, "reward_total_mean": 0.4591677784919739, "reward_meter_mean": 0.4121133089065552, "reward_meter_std": 0.42796626687049866, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.4591677784919739, "reward_total_composite_std": 0.11875561624765396} {"timestamp_utc": "2026-04-13T08:49:22Z", "mode": "train", "global_step": 586, "epoch": 0.05886489201406329, "loss": -0.0906, "grad_norm": 5.4155659675598145, "learning_rate": 8.227272727272728e-06, "num_tokens": 1041915.0, "completions/mean_length": 119.125, "completions/min_length": 57.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 63.000003814697266, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.5183769464492798, "rewards/meter/std": 0.3440532386302948, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9439934492111206, "rewards/repeat_soft/std": 0.08475925028324127, "rewards/judge_quality/mean": 0.45625001192092896, "rewards/judge_quality/std": 0.27102649211883545, "rewards/total_composite/mean": 0.43112117052078247, "rewards/total_composite/std": 0.28696581721305847, "reward": 0.43112117052078247, "reward_std": 0.28696581721305847, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19397583603858948, "sampling/sampling_logp_difference/max": 1.7971467971801758, "sampling/importance_sampling_ratio/min": 0.16577118635177612, "sampling/importance_sampling_ratio/mean": 1.0238008499145508, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4494723081588745, "clip_ratio/low_mean": 0.03227124270051718, "clip_ratio/low_min": 0.03227124270051718, "clip_ratio/high_mean": 0.12011519446969032, "clip_ratio/high_max": 0.12011519446969032, "clip_ratio/region_mean": 0.1523864371702075, "reward_total_mean": 0.43112117052078247, "reward_meter_mean": 0.5183769464492798, "reward_meter_std": 0.3440532386302948, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9439934492111206, "reward_repeat_soft_std": 0.08475925028324127, "reward_judge_quality_mean": 0.45625001192092896, "reward_judge_quality_std": 0.27102649211883545, "reward_total_composite_mean": 0.43112117052078247, "reward_total_composite_std": 0.28696581721305847} {"timestamp_utc": "2026-04-13T08:49:28Z", "mode": "train", "global_step": 587, "epoch": 0.058965344048216975, "loss": 0.0502, "grad_norm": 22.71309471130371, "learning_rate": 8.224242424242425e-06, "num_tokens": 1043235.0, "completions/mean_length": 16.0, "completions/min_length": 14.0, "completions/max_length": 19.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 16.0, "completions/min_terminated_length": 14.0, "completions/max_terminated_length": 19.0, "rewards/meter/mean": 0.7268494367599487, "rewards/meter/std": 0.36402255296707153, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.39249998331069946, "rewards/judge_quality/std": 0.08892211318016052, "rewards/total_composite/mean": 0.4814658463001251, "rewards/total_composite/std": 0.2159145176410675, "reward": 0.4814658463001251, "reward_std": 0.2159145176410675, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18526047468185425, "sampling/sampling_logp_difference/max": 1.2848373651504517, "sampling/importance_sampling_ratio/min": 0.27669557929039, "sampling/importance_sampling_ratio/mean": 1.00920832157135, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5635647252202034, "clip_ratio/low_mean": 0.05330882407724857, "clip_ratio/low_min": 0.05330882407724857, "clip_ratio/high_mean": 0.08062988566234708, "clip_ratio/high_max": 0.08062988566234708, "clip_ratio/region_mean": 0.13393870973959565, "reward_total_mean": 0.4814658463001251, "reward_meter_mean": 0.7268494367599487, "reward_meter_std": 0.36402255296707153, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.39249998331069946, "reward_judge_quality_std": 0.08892211318016052, "reward_total_composite_mean": 0.4814658463001251, "reward_total_composite_std": 0.2159145176410675} {"timestamp_utc": "2026-04-13T08:49:34Z", "mode": "train", "global_step": 588, "epoch": 0.05906579608237067, "loss": 0.0346, "grad_norm": 16.27345085144043, "learning_rate": 8.221212121212122e-06, "num_tokens": 1044755.0, "completions/mean_length": 29.0, "completions/min_length": 24.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.0, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.36250847578048706, "rewards/meter/std": 0.3368504047393799, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9591778516769409, "rewards/repeat_soft/std": 0.037539176642894745, "rewards/judge_quality/mean": 0.6187499761581421, "rewards/judge_quality/std": 0.24976776540279388, "rewards/total_composite/mean": 0.4934357702732086, "rewards/total_composite/std": 0.18997235596179962, "reward": 0.4934357702732086, "reward_std": 0.18997234106063843, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19533571600914001, "sampling/sampling_logp_difference/max": 1.8965137004852295, "sampling/importance_sampling_ratio/min": 0.15009097754955292, "sampling/importance_sampling_ratio/mean": 1.0173364877700806, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.109391801059246, "clip_ratio/low_mean": 0.09431891050189734, "clip_ratio/low_min": 0.09431891050189734, "clip_ratio/high_mean": 0.0604330450296402, "clip_ratio/high_max": 0.0604330450296402, "clip_ratio/region_mean": 0.15475195553153753, "reward_total_mean": 0.4934357702732086, "reward_meter_mean": 0.36250847578048706, "reward_meter_std": 0.3368504047393799, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9591778516769409, "reward_repeat_soft_std": 0.037539176642894745, "reward_judge_quality_mean": 0.6187499761581421, "reward_judge_quality_std": 0.24976776540279388, "reward_total_composite_mean": 0.4934357702732086, "reward_total_composite_std": 0.18997235596179962} {"timestamp_utc": "2026-04-13T08:49:41Z", "mode": "train", "global_step": 589, "epoch": 0.05916624811652436, "loss": -0.0357, "grad_norm": 15.193470001220703, "learning_rate": 8.21818181818182e-06, "num_tokens": 1046080.0, "completions/mean_length": 18.625, "completions/min_length": 16.0, "completions/max_length": 23.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 18.625, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 23.0, "rewards/meter/mean": 0.775652289390564, "rewards/meter/std": 0.3668343424797058, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.5973180532455444, "rewards/total_composite/std": 0.15951067209243774, "reward": 0.5973180532455444, "reward_std": 0.15951068699359894, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10052339732646942, "sampling/sampling_logp_difference/max": 1.0645933151245117, "sampling/importance_sampling_ratio/min": 0.34486809372901917, "sampling/importance_sampling_ratio/mean": 1.0147958993911743, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6124249398708344, "clip_ratio/low_mean": 0.020955882500857115, "clip_ratio/low_min": 0.020955882500857115, "clip_ratio/high_mean": 0.051119307056069374, "clip_ratio/high_max": 0.051119307056069374, "clip_ratio/region_mean": 0.07207518955692649, "reward_total_mean": 0.5973180532455444, "reward_meter_mean": 0.775652289390564, "reward_meter_std": 0.3668343424797058, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.5973180532455444, "reward_total_composite_std": 0.15951067209243774} {"timestamp_utc": "2026-04-13T08:49:47Z", "mode": "train", "global_step": 590, "epoch": 0.05926670015067805, "loss": 0.0741, "grad_norm": 25.138277053833008, "learning_rate": 8.215151515151517e-06, "num_tokens": 1047456.0, "completions/mean_length": 19.0, "completions/min_length": 16.0, "completions/max_length": 22.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 19.0, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 22.0, "rewards/meter/mean": 0.9850324392318726, "rewards/meter/std": 0.014376808889210224, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8506903648376465, "rewards/repeat_soft/std": 0.13034997880458832, "rewards/judge_quality/mean": 0.5862500667572021, "rewards/judge_quality/std": 0.2822834253311157, "rewards/total_composite/mean": 0.7037005424499512, "rewards/total_composite/std": 0.19040586054325104, "reward": 0.7037005424499512, "reward_std": 0.19040584564208984, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19414843618869781, "sampling/sampling_logp_difference/max": 1.3941915035247803, "sampling/importance_sampling_ratio/min": 0.24981577694416046, "sampling/importance_sampling_ratio/mean": 1.0324383974075317, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6125606670975685, "clip_ratio/low_mean": 0.10060129128396511, "clip_ratio/low_min": 0.10060129128396511, "clip_ratio/high_mean": 0.05801083752885461, "clip_ratio/high_max": 0.05801083752885461, "clip_ratio/region_mean": 0.15861212881281972, "reward_total_mean": 0.7037005424499512, "reward_meter_mean": 0.9850324392318726, "reward_meter_std": 0.014376808889210224, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8506903648376465, "reward_repeat_soft_std": 0.13034997880458832, "reward_judge_quality_mean": 0.5862500667572021, "reward_judge_quality_std": 0.2822834253311157, "reward_total_composite_mean": 0.7037005424499512, "reward_total_composite_std": 0.19040586054325104} {"timestamp_utc": "2026-04-13T08:49:54Z", "mode": "train", "global_step": 591, "epoch": 0.05936715218483174, "loss": 0.0241, "grad_norm": 22.737226486206055, "learning_rate": 8.212121212121212e-06, "num_tokens": 1049289.0, "completions/mean_length": 51.125, "completions/min_length": 47.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.125, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.8187070488929749, "rewards/meter/std": 0.1686120331287384, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9087092876434326, "rewards/repeat_soft/std": 0.0625695139169693, "rewards/judge_quality/mean": 0.7074999809265137, "rewards/judge_quality/std": 0.2474873960018158, "rewards/total_composite/mean": 0.7206875085830688, "rewards/total_composite/std": 0.17156639695167542, "reward": 0.7206875085830688, "reward_std": 0.17156638205051422, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1752532571554184, "sampling/sampling_logp_difference/max": 1.7380281686782837, "sampling/importance_sampling_ratio/min": 0.17586684226989746, "sampling/importance_sampling_ratio/mean": 1.0200912952423096, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2432586774230003, "clip_ratio/low_mean": 0.07388498820364475, "clip_ratio/low_min": 0.07388498820364475, "clip_ratio/high_mean": 0.06151804141700268, "clip_ratio/high_max": 0.06151804141700268, "clip_ratio/region_mean": 0.13540302962064743, "reward_total_mean": 0.7206875085830688, "reward_meter_mean": 0.8187070488929749, "reward_meter_std": 0.1686120331287384, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9087092876434326, "reward_repeat_soft_std": 0.0625695139169693, "reward_judge_quality_mean": 0.7074999809265137, "reward_judge_quality_std": 0.2474873960018158, "reward_total_composite_mean": 0.7206875085830688, "reward_total_composite_std": 0.17156639695167542} {"timestamp_utc": "2026-04-13T08:50:06Z", "mode": "train", "global_step": 592, "epoch": 0.059467604218985434, "loss": -0.149, "grad_norm": 3.940671443939209, "learning_rate": 8.20909090909091e-06, "num_tokens": 1051196.0, "completions/mean_length": 125.375, "completions/min_length": 66.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 70.14286041259766, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.8566869497299194, "rewards/meter/std": 0.31202420592308044, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.2651650309562683, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.864543080329895, "rewards/repeat_soft/std": 0.11763402819633484, "rewards/judge_quality/mean": 0.4112499952316284, "rewards/judge_quality/std": 0.1797965168952942, "rewards/total_composite/mean": 0.5127476453781128, "rewards/total_composite/std": 0.2380359172821045, "reward": 0.5127476453781128, "reward_std": 0.2380359172821045, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16044767200946808, "sampling/sampling_logp_difference/max": 1.304091453552246, "sampling/importance_sampling_ratio/min": 0.2714190185070038, "sampling/importance_sampling_ratio/mean": 1.037603735923767, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3022792339324951, "clip_ratio/low_mean": 0.02142857201397419, "clip_ratio/low_min": 0.02142857201397419, "clip_ratio/high_mean": 0.1300391824916005, "clip_ratio/high_max": 0.1300391824916005, "clip_ratio/region_mean": 0.1514677545055747, "reward_total_mean": 0.5127476453781128, "reward_meter_mean": 0.8566869497299194, "reward_meter_std": 0.31202420592308044, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.2651650309562683, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.864543080329895, "reward_repeat_soft_std": 0.11763402819633484, "reward_judge_quality_mean": 0.4112499952316284, "reward_judge_quality_std": 0.1797965168952942, "reward_total_composite_mean": 0.5127476453781128, "reward_total_composite_std": 0.2380359172821045} {"timestamp_utc": "2026-04-13T08:50:13Z", "mode": "train", "global_step": 593, "epoch": 0.05956805625313913, "loss": 0.0705, "grad_norm": 24.125486373901367, "learning_rate": 8.206060606060607e-06, "num_tokens": 1052987.0, "completions/mean_length": 62.875, "completions/min_length": 55.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.875, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.5702065229415894, "rewards/meter/std": 0.37652307748794556, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9546473026275635, "rewards/repeat_soft/std": 0.052884023636579514, "rewards/judge_quality/mean": 0.5024999976158142, "rewards/judge_quality/std": 0.1348809152841568, "rewards/total_composite/mean": 0.5126515030860901, "rewards/total_composite/std": 0.09028740972280502, "reward": 0.5126515030860901, "reward_std": 0.09028739482164383, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19030006229877472, "sampling/sampling_logp_difference/max": 1.6490859985351562, "sampling/importance_sampling_ratio/min": 0.19222553074359894, "sampling/importance_sampling_ratio/mean": 1.0347181558609009, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7659383863210678, "clip_ratio/low_mean": 0.09444891475141048, "clip_ratio/low_min": 0.09444891475141048, "clip_ratio/high_mean": 0.08395612798631191, "clip_ratio/high_max": 0.08395612798631191, "clip_ratio/region_mean": 0.1784050427377224, "reward_total_mean": 0.5126515030860901, "reward_meter_mean": 0.5702065229415894, "reward_meter_std": 0.37652307748794556, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9546473026275635, "reward_repeat_soft_std": 0.052884023636579514, "reward_judge_quality_mean": 0.5024999976158142, "reward_judge_quality_std": 0.1348809152841568, "reward_total_composite_mean": 0.5126515030860901, "reward_total_composite_std": 0.09028740972280502} {"timestamp_utc": "2026-04-13T08:50:20Z", "mode": "train", "global_step": 594, "epoch": 0.05966850828729282, "loss": 0.0226, "grad_norm": 24.708345413208008, "learning_rate": 8.203030303030304e-06, "num_tokens": 1054725.0, "completions/mean_length": 54.25, "completions/min_length": 47.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.25, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.5746891498565674, "rewards/meter/std": 0.33918896317481995, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9928379058837891, "rewards/repeat_soft/std": 0.0032845570240169764, "rewards/judge_quality/mean": 0.5575000047683716, "rewards/judge_quality/std": 0.19955310225486755, "rewards/total_composite/mean": 0.5154507160186768, "rewards/total_composite/std": 0.26690253615379333, "reward": 0.5154507160186768, "reward_std": 0.26690250635147095, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20852459967136383, "sampling/sampling_logp_difference/max": 1.890106201171875, "sampling/importance_sampling_ratio/min": 0.15105575323104858, "sampling/importance_sampling_ratio/mean": 1.010902762413025, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0982920974493027, "clip_ratio/low_mean": 0.09287556447088718, "clip_ratio/low_min": 0.09287556447088718, "clip_ratio/high_mean": 0.09945436753332615, "clip_ratio/high_max": 0.09945436753332615, "clip_ratio/region_mean": 0.19232993200421333, "reward_total_mean": 0.5154507160186768, "reward_meter_mean": 0.5746891498565674, "reward_meter_std": 0.33918896317481995, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9928379058837891, "reward_repeat_soft_std": 0.0032845570240169764, "reward_judge_quality_mean": 0.5575000047683716, "reward_judge_quality_std": 0.19955310225486755, "reward_total_composite_mean": 0.5154507160186768, "reward_total_composite_std": 0.26690253615379333} {"timestamp_utc": "2026-04-13T08:50:26Z", "mode": "train", "global_step": 595, "epoch": 0.05976896032144651, "loss": 0.0578, "grad_norm": 20.776994705200195, "learning_rate": 8.2e-06, "num_tokens": 1056083.0, "completions/mean_length": 19.75, "completions/min_length": 19.0, "completions/max_length": 22.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 19.75, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 22.0, "rewards/meter/mean": 0.7944017648696899, "rewards/meter/std": 0.340792715549469, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.8575000166893005, "rewards/judge_quality/std": 0.176776722073555, "rewards/total_composite/mean": 0.8193592429161072, "rewards/total_composite/std": 0.20397496223449707, "reward": 0.8193592429161072, "reward_std": 0.20397497713565826, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15076832473278046, "sampling/sampling_logp_difference/max": 1.104593276977539, "sampling/importance_sampling_ratio/min": 0.41009634733200073, "sampling/importance_sampling_ratio/mean": 1.0511397123336792, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1086817681789398, "clip_ratio/low_mean": 0.053977273404598236, "clip_ratio/low_min": 0.053977273404598236, "clip_ratio/high_mean": 0.08421052573248744, "clip_ratio/high_max": 0.08421052573248744, "clip_ratio/region_mean": 0.13818779913708568, "reward_total_mean": 0.8193592429161072, "reward_meter_mean": 0.7944017648696899, "reward_meter_std": 0.340792715549469, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.8575000166893005, "reward_judge_quality_std": 0.176776722073555, "reward_total_composite_mean": 0.8193592429161072, "reward_total_composite_std": 0.20397496223449707} {"timestamp_utc": "2026-04-13T08:50:32Z", "mode": "train", "global_step": 596, "epoch": 0.0598694123556002, "loss": 0.0182, "grad_norm": 16.731706619262695, "learning_rate": 8.196969696969698e-06, "num_tokens": 1057927.0, "completions/mean_length": 55.5, "completions/min_length": 47.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.5, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.7734062671661377, "rewards/meter/std": 0.2578055262565613, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9969630837440491, "rewards/repeat_soft/std": 0.002231556223705411, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.47790324687957764, "rewards/total_composite/std": 0.20325395464897156, "reward": 0.47790324687957764, "reward_std": 0.20325393974781036, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19447480142116547, "sampling/sampling_logp_difference/max": 1.691053867340088, "sampling/importance_sampling_ratio/min": 0.1843251734972, "sampling/importance_sampling_ratio/mean": 1.010617971420288, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2968737334012985, "clip_ratio/low_mean": 0.054579597897827625, "clip_ratio/low_min": 0.054579597897827625, "clip_ratio/high_mean": 0.14666733145713806, "clip_ratio/high_max": 0.14666733145713806, "clip_ratio/region_mean": 0.2012469293549657, "reward_total_mean": 0.47790324687957764, "reward_meter_mean": 0.7734062671661377, "reward_meter_std": 0.2578055262565613, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9969630837440491, "reward_repeat_soft_std": 0.002231556223705411, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.47790324687957764, "reward_total_composite_std": 0.20325395464897156} {"timestamp_utc": "2026-04-13T08:50:43Z", "mode": "train", "global_step": 597, "epoch": 0.059969864389753894, "loss": -0.1452, "grad_norm": 2.848376750946045, "learning_rate": 8.193939393939394e-06, "num_tokens": 1059752.0, "completions/mean_length": 121.125, "completions/min_length": 56.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 65.28572082519531, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.7684168815612793, "rewards/meter/std": 0.3329595625400543, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.18600596487522125, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8443343043327332, "rewards/repeat_soft/std": 0.1572190523147583, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.24483230710029602, "rewards/total_composite/mean": 0.5488306879997253, "rewards/total_composite/std": 0.27638494968414307, "reward": 0.5488306879997253, "reward_std": 0.27638494968414307, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14185303449630737, "sampling/sampling_logp_difference/max": 1.3351480960845947, "sampling/importance_sampling_ratio/min": 0.2631192207336426, "sampling/importance_sampling_ratio/mean": 1.0276942253112793, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9028343670070171, "clip_ratio/low_mean": 0.0386904776096344, "clip_ratio/low_min": 0.0386904776096344, "clip_ratio/high_mean": 0.07202116353437304, "clip_ratio/high_max": 0.07202116353437304, "clip_ratio/region_mean": 0.11071164114400744, "reward_total_mean": 0.5488306879997253, "reward_meter_mean": 0.7684168815612793, "reward_meter_std": 0.3329595625400543, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.18600596487522125, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8443343043327332, "reward_repeat_soft_std": 0.1572190523147583, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.24483230710029602, "reward_total_composite_mean": 0.5488306879997253, "reward_total_composite_std": 0.27638494968414307} {"timestamp_utc": "2026-04-13T08:50:49Z", "mode": "train", "global_step": 598, "epoch": 0.06007031642390758, "loss": 0.0874, "grad_norm": 15.479145050048828, "learning_rate": 8.190909090909091e-06, "num_tokens": 1061264.0, "completions/mean_length": 33.0, "completions/min_length": 29.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.0, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.6340094208717346, "rewards/meter/std": 0.44285714626312256, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9069277048110962, "rewards/repeat_soft/std": 0.10584849864244461, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.5099337100982666, "rewards/total_composite/std": 0.11327037215232849, "reward": 0.5099337100982666, "reward_std": 0.1132703647017479, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19748561084270477, "sampling/sampling_logp_difference/max": 1.3864812850952148, "sampling/importance_sampling_ratio/min": 0.2499532848596573, "sampling/importance_sampling_ratio/mean": 1.0355310440063477, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6433321237564087, "clip_ratio/low_mean": 0.10831448249518871, "clip_ratio/low_min": 0.10831448249518871, "clip_ratio/high_mean": 0.09299219027161598, "clip_ratio/high_max": 0.09299219027161598, "clip_ratio/region_mean": 0.2013066727668047, "reward_total_mean": 0.5099337100982666, "reward_meter_mean": 0.6340094208717346, "reward_meter_std": 0.44285714626312256, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9069277048110962, "reward_repeat_soft_std": 0.10584849864244461, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.5099337100982666, "reward_total_composite_std": 0.11327037215232849} {"timestamp_utc": "2026-04-13T08:51:01Z", "mode": "train", "global_step": 599, "epoch": 0.060170768458061276, "loss": -0.1053, "grad_norm": 2.855733871459961, "learning_rate": 8.187878787878788e-06, "num_tokens": 1062902.0, "completions/mean_length": 98.75, "completions/min_length": 35.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 39.71428680419922, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.5551939010620117, "rewards/meter/std": 0.4094555675983429, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.972247302532196, "rewards/repeat_soft/std": 0.04996117576956749, "rewards/judge_quality/mean": 0.42250001430511475, "rewards/judge_quality/std": 0.24423348903656006, "rewards/total_composite/mean": 0.44813311100006104, "rewards/total_composite/std": 0.20157425105571747, "reward": 0.44813311100006104, "reward_std": 0.20157423615455627, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1727154403924942, "sampling/sampling_logp_difference/max": 1.759019374847412, "sampling/importance_sampling_ratio/min": 0.17221365869045258, "sampling/importance_sampling_ratio/mean": 1.0211820602416992, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3214404955506325, "clip_ratio/low_mean": 0.02302631549537182, "clip_ratio/low_min": 0.02302631549537182, "clip_ratio/high_mean": 0.12357525061815977, "clip_ratio/high_max": 0.12357525061815977, "clip_ratio/region_mean": 0.1466015661135316, "reward_total_mean": 0.44813311100006104, "reward_meter_mean": 0.5551939010620117, "reward_meter_std": 0.4094555675983429, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.972247302532196, "reward_repeat_soft_std": 0.04996117576956749, "reward_judge_quality_mean": 0.42250001430511475, "reward_judge_quality_std": 0.24423348903656006, "reward_total_composite_mean": 0.44813311100006104, "reward_total_composite_std": 0.20157425105571747} {"timestamp_utc": "2026-04-13T08:51:07Z", "mode": "train", "global_step": 600, "epoch": 0.06027122049221497, "loss": 0.1465, "grad_norm": 15.054155349731445, "learning_rate": 8.184848484848486e-06, "num_tokens": 1064284.0, "completions/mean_length": 23.75, "completions/min_length": 22.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.75, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.950312077999115, "rewards/meter/std": 0.06284411251544952, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9781278371810913, "rewards/repeat_soft/std": 0.014061374589800835, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.6061543822288513, "rewards/total_composite/std": 0.018617939203977585, "reward": 0.6061543822288513, "reward_std": 0.018617935478687286, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06382521986961365, "sampling/sampling_logp_difference/max": 1.6088030338287354, "sampling/importance_sampling_ratio/min": 0.2001270055770874, "sampling/importance_sampling_ratio/mean": 1.0066806077957153, "sampling/importance_sampling_ratio/max": 1.8192110061645508, "entropy": 0.23449560813605785, "clip_ratio/low_mean": 0.018939394503831863, "clip_ratio/low_min": 0.018939394503831863, "clip_ratio/high_mean": 0.0437870561145246, "clip_ratio/high_max": 0.0437870561145246, "clip_ratio/region_mean": 0.06272645061835647, "reward_total_mean": 0.6061543822288513, "reward_meter_mean": 0.950312077999115, "reward_meter_std": 0.06284411251544952, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9781278371810913, "reward_repeat_soft_std": 0.014061374589800835, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.6061543822288513, "reward_total_composite_std": 0.018617939203977585} {"timestamp_utc": "2026-04-13T08:52:03Z", "mode": "eval", "global_step": 600, "epoch": 0.06027122049221497, "eval_loss": NaN, "eval_runtime": 56.8265, "eval_samples_per_second": 1.408, "eval_steps_per_second": 0.176, "eval_num_tokens": 1064284.0, "eval_completions/mean_length": 84.6, "eval_completions/min_length": 30.9, "eval_completions/max_length": 260.2, "eval_completions/clipped_ratio": 0.0625, "eval_completions/mean_terminated_length": 56.774406051635744, "eval_completions/min_terminated_length": 30.9, "eval_completions/max_terminated_length": 97.4, "eval_rewards/meter/mean": 0.5933304131031036, "eval_rewards/meter/std": 0.3523423582315445, "eval_rewards/count_adherence/mean": 0.9745833277702332, "eval_rewards/count_adherence/std": 0.07188918963074684, "eval_rewards/hard_gate/mean": 0.9125, "eval_rewards/hard_gate/std": 0.19864802658557892, "eval_rewards/repeat_soft/mean": 0.9164949417114258, "eval_rewards/repeat_soft/std": 0.11432811766862869, "eval_rewards/judge_quality/mean": 0.5162499994039536, "eval_rewards/judge_quality/std": 0.23027257323265077, "eval_rewards/total_composite/mean": 0.5080294042825699, "eval_rewards/total_composite/std": 0.20953540951013566, "eval_reward": 0.5080294042825699, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.09997781440615654, "eval_sampling/sampling_logp_difference/max": 0.9178894281387329, "eval_sampling/importance_sampling_ratio/min": 0.403290730714798, "eval_sampling/importance_sampling_ratio/mean": 1.0318517208099365, "eval_sampling/importance_sampling_ratio/max": 1.5025453090667724, "eval_entropy": 1.2897954106330871, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5080294042825699, "eval_reward_meter_mean": 0.5933304131031036, "eval_reward_meter_std": 0.3523423582315445, "eval_reward_count_adherence_mean": 0.9745833277702332, "eval_reward_count_adherence_std": 0.07188918963074684, "eval_reward_hard_gate_mean": 0.9125, "eval_reward_hard_gate_std": 0.19864802658557892, "eval_reward_repeat_soft_mean": 0.9164949417114258, "eval_reward_repeat_soft_std": 0.11432811766862869, "eval_reward_judge_quality_mean": 0.5162499994039536, "eval_reward_judge_quality_std": 0.23027257323265077, "eval_reward_total_composite_mean": 0.5080294042825699, "eval_reward_total_composite_std": 0.20953540951013566} {"timestamp_utc": "2026-04-13T08:52:13Z", "mode": "train", "global_step": 601, "epoch": 0.06037167252636866, "loss": 0.0111, "grad_norm": 9.83389663696289, "learning_rate": 8.181818181818183e-06, "num_tokens": 1066702.0, "completions/mean_length": 69.25, "completions/min_length": 63.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.25, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.9074545502662659, "rewards/meter/std": 0.14883460104465485, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9192219972610474, "rewards/repeat_soft/std": 0.0344962514936924, "rewards/judge_quality/mean": 0.4987499713897705, "rewards/judge_quality/std": 0.21357084810733795, "rewards/total_composite/mean": 0.6348594427108765, "rewards/total_composite/std": 0.16552099585533142, "reward": 0.6348594427108765, "reward_std": 0.16552099585533142, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1714843213558197, "sampling/sampling_logp_difference/max": 1.6988840103149414, "sampling/importance_sampling_ratio/min": 0.1828875094652176, "sampling/importance_sampling_ratio/mean": 1.02165949344635, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6118394806981087, "clip_ratio/low_mean": 0.10102782025933266, "clip_ratio/low_min": 0.10102782025933266, "clip_ratio/high_mean": 0.040490858256816864, "clip_ratio/high_max": 0.040490858256816864, "clip_ratio/region_mean": 0.14151867851614952, "reward_total_mean": 0.6348594427108765, "reward_meter_mean": 0.9074545502662659, "reward_meter_std": 0.14883460104465485, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9192219972610474, "reward_repeat_soft_std": 0.0344962514936924, "reward_judge_quality_mean": 0.4987499713897705, "reward_judge_quality_std": 0.21357084810733795, "reward_total_composite_mean": 0.6348594427108765, "reward_total_composite_std": 0.16552099585533142} {"timestamp_utc": "2026-04-13T08:52:20Z", "mode": "train", "global_step": 602, "epoch": 0.06047212456052235, "loss": 0.0697, "grad_norm": 9.658855438232422, "learning_rate": 8.17878787878788e-06, "num_tokens": 1069606.0, "completions/mean_length": 124.0, "completions/min_length": 97.0, "completions/max_length": 164.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 124.0, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 164.0, "rewards/meter/mean": 0.4799366593360901, "rewards/meter/std": 0.302285373210907, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.08625820279121399, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.918707549571991, "rewards/repeat_soft/std": 0.09461511671543121, "rewards/judge_quality/mean": 0.4362499713897705, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.46457159519195557, "rewards/total_composite/std": 0.0959501639008522, "reward": 0.46457159519195557, "reward_std": 0.0959501564502716, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19479040801525116, "sampling/sampling_logp_difference/max": 2.191174030303955, "sampling/importance_sampling_ratio/min": 0.11178543418645859, "sampling/importance_sampling_ratio/mean": 1.0257471799850464, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7584725245833397, "clip_ratio/low_mean": 0.10992318764328957, "clip_ratio/low_min": 0.10992318764328957, "clip_ratio/high_mean": 0.04259421583265066, "clip_ratio/high_max": 0.04259421583265066, "clip_ratio/region_mean": 0.15251740347594023, "reward_total_mean": 0.46457159519195557, "reward_meter_mean": 0.4799366593360901, "reward_meter_std": 0.302285373210907, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.08625820279121399, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.918707549571991, "reward_repeat_soft_std": 0.09461511671543121, "reward_judge_quality_mean": 0.4362499713897705, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.46457159519195557, "reward_total_composite_std": 0.0959501639008522} {"timestamp_utc": "2026-04-13T08:52:32Z", "mode": "train", "global_step": 603, "epoch": 0.06057257659467604, "loss": -0.2125, "grad_norm": 2.0980961322784424, "learning_rate": 8.175757575757577e-06, "num_tokens": 1071893.0, "completions/mean_length": 271.875, "completions/min_length": 110.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 127.80000305175781, "completions/min_terminated_length": 110.0, "completions/max_terminated_length": 158.0, "rewards/meter/mean": 0.726004421710968, "rewards/meter/std": 0.292203813791275, "rewards/count_adherence/mean": 0.8333333730697632, "rewards/count_adherence/std": 0.2519763112068176, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.8333075046539307, "rewards/repeat_soft/std": 0.13805323839187622, "rewards/judge_quality/mean": 0.22999998927116394, "rewards/judge_quality/std": 0.1437259316444397, "rewards/total_composite/mean": 0.2765088677406311, "rewards/total_composite/std": 0.23182496428489685, "reward": 0.2765088677406311, "reward_std": 0.23182496428489685, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13859088718891144, "sampling/sampling_logp_difference/max": 1.422471046447754, "sampling/importance_sampling_ratio/min": 0.2411174774169922, "sampling/importance_sampling_ratio/mean": 1.0195611715316772, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7712185755372047, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0950105395168066, "clip_ratio/high_max": 0.0950105395168066, "clip_ratio/region_mean": 0.0950105395168066, "reward_total_mean": 0.2765088677406311, "reward_meter_mean": 0.726004421710968, "reward_meter_std": 0.292203813791275, "reward_count_adherence_mean": 0.8333333730697632, "reward_count_adherence_std": 0.2519763112068176, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.8333075046539307, "reward_repeat_soft_std": 0.13805323839187622, "reward_judge_quality_mean": 0.22999998927116394, "reward_judge_quality_std": 0.1437259316444397, "reward_total_composite_mean": 0.2765088677406311, "reward_total_composite_std": 0.23182496428489685} {"timestamp_utc": "2026-04-13T08:52:43Z", "mode": "train", "global_step": 604, "epoch": 0.060673028628829735, "loss": -0.074, "grad_norm": 1.7580572366714478, "learning_rate": 8.172727272727273e-06, "num_tokens": 1073313.0, "completions/mean_length": 150.5, "completions/min_length": 27.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 30.0, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.8189277648925781, "rewards/meter/std": 0.3430798351764679, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9567056894302368, "rewards/repeat_soft/std": 0.05685315281152725, "rewards/judge_quality/mean": 0.3137499988079071, "rewards/judge_quality/std": 0.17492344975471497, "rewards/total_composite/mean": 0.44049832224845886, "rewards/total_composite/std": 0.2767917811870575, "reward": 0.44049832224845886, "reward_std": 0.2767917513847351, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18948166072368622, "sampling/sampling_logp_difference/max": 1.9374151229858398, "sampling/importance_sampling_ratio/min": 0.1440758854150772, "sampling/importance_sampling_ratio/mean": 1.0077388286590576, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3183616027235985, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.11339256633073092, "clip_ratio/high_max": 0.11339256633073092, "clip_ratio/region_mean": 0.11339256633073092, "reward_total_mean": 0.44049832224845886, "reward_meter_mean": 0.8189277648925781, "reward_meter_std": 0.3430798351764679, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9567056894302368, "reward_repeat_soft_std": 0.05685315281152725, "reward_judge_quality_mean": 0.3137499988079071, "reward_judge_quality_std": 0.17492344975471497, "reward_total_composite_mean": 0.44049832224845886, "reward_total_composite_std": 0.2767917811870575} {"timestamp_utc": "2026-04-13T08:52:49Z", "mode": "train", "global_step": 605, "epoch": 0.06077348066298342, "loss": 0.0357, "grad_norm": 15.67078685760498, "learning_rate": 8.16969696969697e-06, "num_tokens": 1074844.0, "completions/mean_length": 30.375, "completions/min_length": 26.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.375, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9137496948242188, "rewards/meter/std": 0.09076227247714996, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8560148477554321, "rewards/repeat_soft/std": 0.05153679475188255, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.622412919998169, "rewards/total_composite/std": 0.12228536605834961, "reward": 0.622412919998169, "reward_std": 0.12228536605834961, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17435169219970703, "sampling/sampling_logp_difference/max": 2.398494005203247, "sampling/importance_sampling_ratio/min": 0.09085467457771301, "sampling/importance_sampling_ratio/mean": 1.010645866394043, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.181313119828701, "clip_ratio/low_mean": 0.11573188286274672, "clip_ratio/low_min": 0.11573188286274672, "clip_ratio/high_mean": 0.02083333395421505, "clip_ratio/high_max": 0.02083333395421505, "clip_ratio/region_mean": 0.13656521681696177, "reward_total_mean": 0.622412919998169, "reward_meter_mean": 0.9137496948242188, "reward_meter_std": 0.09076227247714996, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8560148477554321, "reward_repeat_soft_std": 0.05153679475188255, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.622412919998169, "reward_total_composite_std": 0.12228536605834961} {"timestamp_utc": "2026-04-13T08:53:00Z", "mode": "train", "global_step": 606, "epoch": 0.06087393269713712, "loss": -0.084, "grad_norm": 4.847084999084473, "learning_rate": 8.166666666666668e-06, "num_tokens": 1076268.0, "completions/mean_length": 92.0, "completions/min_length": 27.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.29814398288726807, "rewards/meter/std": 0.35805410146713257, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9650765657424927, "rewards/repeat_soft/std": 0.03833828493952751, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.23445606231689453, "rewards/total_composite/mean": 0.37954479455947876, "rewards/total_composite/std": 0.1818016916513443, "reward": 0.37954479455947876, "reward_std": 0.1818016916513443, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1880868524312973, "sampling/sampling_logp_difference/max": 1.379664421081543, "sampling/importance_sampling_ratio/min": 0.2516629993915558, "sampling/importance_sampling_ratio/mean": 1.0463781356811523, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.129436418414116, "clip_ratio/low_mean": 0.08789344504475594, "clip_ratio/low_min": 0.08789344504475594, "clip_ratio/high_mean": 0.039434524485841393, "clip_ratio/high_max": 0.039434524485841393, "clip_ratio/region_mean": 0.12732796953059733, "reward_total_mean": 0.37954479455947876, "reward_meter_mean": 0.29814398288726807, "reward_meter_std": 0.35805410146713257, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9650765657424927, "reward_repeat_soft_std": 0.03833828493952751, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.23445606231689453, "reward_total_composite_mean": 0.37954479455947876, "reward_total_composite_std": 0.1818016916513443} {"timestamp_utc": "2026-04-13T08:53:13Z", "mode": "train", "global_step": 607, "epoch": 0.060974384731290805, "loss": -0.1326, "grad_norm": 2.402979612350464, "learning_rate": 8.163636363636365e-06, "num_tokens": 1077917.0, "completions/mean_length": 111.125, "completions/min_length": 45.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 53.85714340209961, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.6160660982131958, "rewards/meter/std": 0.28426313400268555, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9823377132415771, "rewards/repeat_soft/std": 0.02172774076461792, "rewards/judge_quality/mean": 0.5062500238418579, "rewards/judge_quality/std": 0.24991071224212646, "rewards/total_composite/mean": 0.5163519382476807, "rewards/total_composite/std": 0.21996428072452545, "reward": 0.5163519382476807, "reward_std": 0.21996428072452545, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16341672837734222, "sampling/sampling_logp_difference/max": 2.2435765266418457, "sampling/importance_sampling_ratio/min": 0.10607843846082687, "sampling/importance_sampling_ratio/mean": 1.002920389175415, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7133377343416214, "clip_ratio/low_mean": 0.016509434208273888, "clip_ratio/low_min": 0.016509434208273888, "clip_ratio/high_mean": 0.11188606917858124, "clip_ratio/high_max": 0.11188606917858124, "clip_ratio/region_mean": 0.12839550338685513, "reward_total_mean": 0.5163519382476807, "reward_meter_mean": 0.6160660982131958, "reward_meter_std": 0.28426313400268555, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9823377132415771, "reward_repeat_soft_std": 0.02172774076461792, "reward_judge_quality_mean": 0.5062500238418579, "reward_judge_quality_std": 0.24991071224212646, "reward_total_composite_mean": 0.5163519382476807, "reward_total_composite_std": 0.21996428072452545} {"timestamp_utc": "2026-04-13T08:53:25Z", "mode": "train", "global_step": 608, "epoch": 0.0610748367654445, "loss": -0.0922, "grad_norm": 1.6431916952133179, "learning_rate": 8.16060606060606e-06, "num_tokens": 1079286.0, "completions/mean_length": 88.125, "completions/min_length": 25.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 27.571430206298828, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.78997403383255, "rewards/meter/std": 0.3272418975830078, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9565540552139282, "rewards/repeat_soft/std": 0.04775778576731682, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.5152409076690674, "rewards/total_composite/std": 0.20895171165466309, "reward": 0.5152409076690674, "reward_std": 0.20895171165466309, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1754993200302124, "sampling/sampling_logp_difference/max": 2.666184663772583, "sampling/importance_sampling_ratio/min": 0.06951694935560226, "sampling/importance_sampling_ratio/mean": 1.0019959211349487, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0098569318652153, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.17423375509679317, "clip_ratio/high_max": 0.17423375509679317, "clip_ratio/region_mean": 0.17423375509679317, "reward_total_mean": 0.5152409076690674, "reward_meter_mean": 0.78997403383255, "reward_meter_std": 0.3272418975830078, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9565540552139282, "reward_repeat_soft_std": 0.04775778576731682, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.5152409076690674, "reward_total_composite_std": 0.20895171165466309} {"timestamp_utc": "2026-04-13T08:53:36Z", "mode": "train", "global_step": 609, "epoch": 0.061175288799598194, "loss": -0.0935, "grad_norm": 3.18972110748291, "learning_rate": 8.15757575757576e-06, "num_tokens": 1080771.0, "completions/mean_length": 93.625, "completions/min_length": 29.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 33.85714340209961, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.5269536972045898, "rewards/meter/std": 0.34882891178131104, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9965314269065857, "rewards/repeat_soft/std": 0.00570334866642952, "rewards/judge_quality/mean": 0.3812500238418579, "rewards/judge_quality/std": 0.13452960550785065, "rewards/total_composite/mean": 0.4282902479171753, "rewards/total_composite/std": 0.1965111643075943, "reward": 0.4282902479171753, "reward_std": 0.1965111643075943, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18389181792736053, "sampling/sampling_logp_difference/max": 1.4303655624389648, "sampling/importance_sampling_ratio/min": 0.239221453666687, "sampling/importance_sampling_ratio/mean": 1.0592796802520752, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5589155703783035, "clip_ratio/low_mean": 0.05129310209304094, "clip_ratio/low_min": 0.05129310209304094, "clip_ratio/high_mean": 0.09888686705380678, "clip_ratio/high_max": 0.09888686705380678, "clip_ratio/region_mean": 0.15017996914684772, "reward_total_mean": 0.4282902479171753, "reward_meter_mean": 0.5269536972045898, "reward_meter_std": 0.34882891178131104, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9965314269065857, "reward_repeat_soft_std": 0.00570334866642952, "reward_judge_quality_mean": 0.3812500238418579, "reward_judge_quality_std": 0.13452960550785065, "reward_total_composite_mean": 0.4282902479171753, "reward_total_composite_std": 0.1965111643075943} {"timestamp_utc": "2026-04-13T08:53:48Z", "mode": "train", "global_step": 610, "epoch": 0.06127574083375188, "loss": -0.0698, "grad_norm": 1.857463002204895, "learning_rate": 8.154545454545455e-06, "num_tokens": 1082119.0, "completions/mean_length": 150.5, "completions/min_length": 21.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 30.0, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.23472517728805542, "rewards/meter/std": 0.2359900027513504, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.8774616718292236, "rewards/repeat_soft/std": 0.08078308403491974, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.3728941082954407, "rewards/total_composite/mean": 0.32463350892066956, "rewards/total_composite/std": 0.22264796495437622, "reward": 0.32463350892066956, "reward_std": 0.22264795005321503, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18068988621234894, "sampling/sampling_logp_difference/max": 1.5548620223999023, "sampling/importance_sampling_ratio/min": 0.21121850609779358, "sampling/importance_sampling_ratio/mean": 1.006516456604004, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7865127325057983, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.11486065573990345, "clip_ratio/high_max": 0.11486065573990345, "clip_ratio/region_mean": 0.11486065573990345, "reward_total_mean": 0.32463350892066956, "reward_meter_mean": 0.23472517728805542, "reward_meter_std": 0.2359900027513504, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.8774616718292236, "reward_repeat_soft_std": 0.08078308403491974, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.3728941082954407, "reward_total_composite_mean": 0.32463350892066956, "reward_total_composite_std": 0.22264796495437622} {"timestamp_utc": "2026-04-13T08:53:59Z", "mode": "train", "global_step": 611, "epoch": 0.06137619286790558, "loss": -0.0851, "grad_norm": 3.469688653945923, "learning_rate": 8.151515151515152e-06, "num_tokens": 1083519.0, "completions/mean_length": 94.0, "completions/min_length": 32.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 34.28571701049805, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.6625721454620361, "rewards/meter/std": 0.40199291706085205, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9902437925338745, "rewards/repeat_soft/std": 0.00864100456237793, "rewards/judge_quality/mean": 0.6862499713897705, "rewards/judge_quality/std": 0.34221702814102173, "rewards/total_composite/mean": 0.6244262456893921, "rewards/total_composite/std": 0.32269027829170227, "reward": 0.6244262456893921, "reward_std": 0.32269027829170227, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19811153411865234, "sampling/sampling_logp_difference/max": 1.5171453952789307, "sampling/importance_sampling_ratio/min": 0.2193371206521988, "sampling/importance_sampling_ratio/mean": 0.9901535511016846, "sampling/importance_sampling_ratio/max": 1.8197382688522339, "entropy": 1.529385268688202, "clip_ratio/low_mean": 0.051674107322469354, "clip_ratio/low_min": 0.051674107322469354, "clip_ratio/high_mean": 0.07405303046107292, "clip_ratio/high_max": 0.07405303046107292, "clip_ratio/region_mean": 0.12572713778354228, "reward_total_mean": 0.6244262456893921, "reward_meter_mean": 0.6625721454620361, "reward_meter_std": 0.40199291706085205, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9902437925338745, "reward_repeat_soft_std": 0.00864100456237793, "reward_judge_quality_mean": 0.6862499713897705, "reward_judge_quality_std": 0.34221702814102173, "reward_total_composite_mean": 0.6244262456893921, "reward_total_composite_std": 0.32269027829170227} {"timestamp_utc": "2026-04-13T08:54:11Z", "mode": "train", "global_step": 612, "epoch": 0.061476644902059265, "loss": -0.1151, "grad_norm": 2.0957794189453125, "learning_rate": 8.14848484848485e-06, "num_tokens": 1085081.0, "completions/mean_length": 166.25, "completions/min_length": 43.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 51.0, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.46912112832069397, "rewards/meter/std": 0.415269672870636, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9899865388870239, "rewards/repeat_soft/std": 0.016494235023856163, "rewards/judge_quality/mean": 0.38999998569488525, "rewards/judge_quality/std": 0.27166154980659485, "rewards/total_composite/mean": 0.3901972770690918, "rewards/total_composite/std": 0.254092812538147, "reward": 0.3901972770690918, "reward_std": 0.254092812538147, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18144994974136353, "sampling/sampling_logp_difference/max": 1.5548648834228516, "sampling/importance_sampling_ratio/min": 0.21121792495250702, "sampling/importance_sampling_ratio/mean": 1.044118881225586, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3020441830158234, "clip_ratio/low_mean": 0.013020833022892475, "clip_ratio/low_min": 0.013020833022892475, "clip_ratio/high_mean": 0.10932030715048313, "clip_ratio/high_max": 0.10932030715048313, "clip_ratio/region_mean": 0.1223411401733756, "reward_total_mean": 0.3901972770690918, "reward_meter_mean": 0.46912112832069397, "reward_meter_std": 0.415269672870636, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9899865388870239, "reward_repeat_soft_std": 0.016494235023856163, "reward_judge_quality_mean": 0.38999998569488525, "reward_judge_quality_std": 0.27166154980659485, "reward_total_composite_mean": 0.3901972770690918, "reward_total_composite_std": 0.254092812538147} {"timestamp_utc": "2026-04-13T08:54:17Z", "mode": "train", "global_step": 613, "epoch": 0.06157709693621296, "loss": -0.1371, "grad_norm": 21.545236587524414, "learning_rate": 8.145454545454547e-06, "num_tokens": 1086456.0, "completions/mean_length": 14.875, "completions/min_length": 9.0, "completions/max_length": 20.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 14.875, "completions/min_terminated_length": 9.0, "completions/max_terminated_length": 20.0, "rewards/meter/mean": 0.2173997163772583, "rewards/meter/std": 0.34037137031555176, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.5175492167472839, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.974461019039154, "rewards/repeat_soft/std": 0.021496828645467758, "rewards/judge_quality/mean": 0.4137499928474426, "rewards/judge_quality/std": 0.23366262018680573, "rewards/total_composite/mean": 0.28171297907829285, "rewards/total_composite/std": 0.24723458290100098, "reward": 0.28171297907829285, "reward_std": 0.2472345530986786, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20753099024295807, "sampling/sampling_logp_difference/max": 1.3933192491531372, "sampling/importance_sampling_ratio/min": 0.24824993312358856, "sampling/importance_sampling_ratio/mean": 0.975199818611145, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1872248575091362, "clip_ratio/low_mean": 0.02291666716337204, "clip_ratio/low_min": 0.02291666716337204, "clip_ratio/high_mean": 0.08686145581305027, "clip_ratio/high_max": 0.08686145581305027, "clip_ratio/region_mean": 0.10977812297642231, "reward_total_mean": 0.28171297907829285, "reward_meter_mean": 0.2173997163772583, "reward_meter_std": 0.34037137031555176, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.5175492167472839, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.974461019039154, "reward_repeat_soft_std": 0.021496828645467758, "reward_judge_quality_mean": 0.4137499928474426, "reward_judge_quality_std": 0.23366262018680573, "reward_total_composite_mean": 0.28171297907829285, "reward_total_composite_std": 0.24723458290100098} {"timestamp_utc": "2026-04-13T08:54:23Z", "mode": "train", "global_step": 614, "epoch": 0.06167754897036665, "loss": 0.0372, "grad_norm": 31.44270133972168, "learning_rate": 8.142424242424242e-06, "num_tokens": 1087872.0, "completions/mean_length": 30.0, "completions/min_length": 27.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.0, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9214865565299988, "rewards/meter/std": 0.1623855084180832, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9567240476608276, "rewards/repeat_soft/std": 0.03225353732705116, "rewards/judge_quality/mean": 0.4724999964237213, "rewards/judge_quality/std": 0.1011011004447937, "rewards/total_composite/mean": 0.6277033090591431, "rewards/total_composite/std": 0.08398663997650146, "reward": 0.6277033090591431, "reward_std": 0.08398663997650146, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17209674417972565, "sampling/sampling_logp_difference/max": 1.3161673545837402, "sampling/importance_sampling_ratio/min": 0.2681611180305481, "sampling/importance_sampling_ratio/mean": 1.024430513381958, "sampling/importance_sampling_ratio/max": 1.84481942653656, "entropy": 1.3749855011701584, "clip_ratio/low_mean": 0.11485749203711748, "clip_ratio/low_min": 0.11485749203711748, "clip_ratio/high_mean": 0.03750000149011612, "clip_ratio/high_max": 0.03750000149011612, "clip_ratio/region_mean": 0.1523574935272336, "reward_total_mean": 0.6277033090591431, "reward_meter_mean": 0.9214865565299988, "reward_meter_std": 0.1623855084180832, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9567240476608276, "reward_repeat_soft_std": 0.03225353732705116, "reward_judge_quality_mean": 0.4724999964237213, "reward_judge_quality_std": 0.1011011004447937, "reward_total_composite_mean": 0.6277033090591431, "reward_total_composite_std": 0.08398663997650146} {"timestamp_utc": "2026-04-13T08:54:35Z", "mode": "train", "global_step": 615, "epoch": 0.06177800100452034, "loss": -0.1034, "grad_norm": 1.75832998752594, "learning_rate": 8.139393939393941e-06, "num_tokens": 1089672.0, "completions/mean_length": 289.0, "completions/min_length": 63.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.5, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.7459532022476196, "rewards/meter/std": 0.32093924283981323, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 0.5, "rewards/hard_gate/std": 0.5345224738121033, "rewards/repeat_soft/mean": 0.893631637096405, "rewards/repeat_soft/std": 0.12420381605625153, "rewards/judge_quality/mean": 0.23499998450279236, "rewards/judge_quality/std": 0.19777332246303558, "rewards/total_composite/mean": 0.29106974601745605, "rewards/total_composite/std": 0.31163278222084045, "reward": 0.29106974601745605, "reward_std": 0.31163278222084045, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18158908188343048, "sampling/sampling_logp_difference/max": 1.7608652114868164, "sampling/importance_sampling_ratio/min": 0.1718960702419281, "sampling/importance_sampling_ratio/mean": 1.0098347663879395, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7075674682855606, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.08472961373627186, "clip_ratio/high_max": 0.08472961373627186, "clip_ratio/region_mean": 0.08472961373627186, "reward_total_mean": 0.29106974601745605, "reward_meter_mean": 0.7459532022476196, "reward_meter_std": 0.32093924283981323, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 0.5, "reward_hard_gate_std": 0.5345224738121033, "reward_repeat_soft_mean": 0.893631637096405, "reward_repeat_soft_std": 0.12420381605625153, "reward_judge_quality_mean": 0.23499998450279236, "reward_judge_quality_std": 0.19777332246303558, "reward_total_composite_mean": 0.29106974601745605, "reward_total_composite_std": 0.31163278222084045} {"timestamp_utc": "2026-04-13T08:54:41Z", "mode": "train", "global_step": 616, "epoch": 0.061878453038674036, "loss": 0.0415, "grad_norm": 19.410985946655273, "learning_rate": 8.136363636363637e-06, "num_tokens": 1091114.0, "completions/mean_length": 34.25, "completions/min_length": 27.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.25, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.398084819316864, "rewards/meter/std": 0.3238059878349304, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9631776809692383, "rewards/repeat_soft/std": 0.0329229012131691, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.4093209207057953, "rewards/total_composite/std": 0.18217498064041138, "reward": 0.4093209207057953, "reward_std": 0.18217498064041138, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17289279401302338, "sampling/sampling_logp_difference/max": 1.4119362831115723, "sampling/importance_sampling_ratio/min": 0.2436710149049759, "sampling/importance_sampling_ratio/mean": 1.024407982826233, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2369627207517624, "clip_ratio/low_mean": 0.06231678184121847, "clip_ratio/low_min": 0.06231678184121847, "clip_ratio/high_mean": 0.09158502984791994, "clip_ratio/high_max": 0.09158502984791994, "clip_ratio/region_mean": 0.1539018116891384, "reward_total_mean": 0.4093209207057953, "reward_meter_mean": 0.398084819316864, "reward_meter_std": 0.3238059878349304, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9631776809692383, "reward_repeat_soft_std": 0.0329229012131691, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.4093209207057953, "reward_total_composite_std": 0.18217498064041138} {"timestamp_utc": "2026-04-13T08:54:47Z", "mode": "train", "global_step": 617, "epoch": 0.061978905072827724, "loss": 0.1614, "grad_norm": 20.967464447021484, "learning_rate": 8.133333333333334e-06, "num_tokens": 1092504.0, "completions/mean_length": 25.75, "completions/min_length": 19.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.75, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.4370259642601013, "rewards/meter/std": 0.3252226412296295, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9890512228012085, "rewards/repeat_soft/std": 0.011672230437397957, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.4696791172027588, "rewards/total_composite/std": 0.08837082237005234, "reward": 0.4696791172027588, "reward_std": 0.08837081491947174, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15416434407234192, "sampling/sampling_logp_difference/max": 1.711958885192871, "sampling/importance_sampling_ratio/min": 0.18051183223724365, "sampling/importance_sampling_ratio/mean": 0.9849514365196228, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7167652025818825, "clip_ratio/low_mean": 0.10505807586014271, "clip_ratio/low_min": 0.10505807586014271, "clip_ratio/high_mean": 0.07016194332391024, "clip_ratio/high_max": 0.07016194332391024, "clip_ratio/region_mean": 0.17522001918405294, "reward_total_mean": 0.4696791172027588, "reward_meter_mean": 0.4370259642601013, "reward_meter_std": 0.3252226412296295, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9890512228012085, "reward_repeat_soft_std": 0.011672230437397957, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.4696791172027588, "reward_total_composite_std": 0.08837082237005234} {"timestamp_utc": "2026-04-13T08:54:58Z", "mode": "train", "global_step": 618, "epoch": 0.06207935710698142, "loss": -0.1143, "grad_norm": 3.0757648944854736, "learning_rate": 8.130303030303031e-06, "num_tokens": 1094125.0, "completions/mean_length": 103.625, "completions/min_length": 41.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 45.28571701049805, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.7123228311538696, "rewards/meter/std": 0.4260020852088928, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8826066255569458, "rewards/repeat_soft/std": 0.1023331731557846, "rewards/judge_quality/mean": 0.2849999964237213, "rewards/judge_quality/std": 0.15390163660049438, "rewards/total_composite/mean": 0.446541428565979, "rewards/total_composite/std": 0.20337434113025665, "reward": 0.446541428565979, "reward_std": 0.20337434113025665, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18212325870990753, "sampling/sampling_logp_difference/max": 1.821512222290039, "sampling/importance_sampling_ratio/min": 0.1617809236049652, "sampling/importance_sampling_ratio/mean": 1.0186419486999512, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5973176509141922, "clip_ratio/low_mean": 0.03890810161828995, "clip_ratio/low_min": 0.03890810161828995, "clip_ratio/high_mean": 0.131703345105052, "clip_ratio/high_max": 0.131703345105052, "clip_ratio/region_mean": 0.17061144672334194, "reward_total_mean": 0.446541428565979, "reward_meter_mean": 0.7123228311538696, "reward_meter_std": 0.4260020852088928, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8826066255569458, "reward_repeat_soft_std": 0.1023331731557846, "reward_judge_quality_mean": 0.2849999964237213, "reward_judge_quality_std": 0.15390163660049438, "reward_total_composite_mean": 0.446541428565979, "reward_total_composite_std": 0.20337434113025665} {"timestamp_utc": "2026-04-13T08:55:05Z", "mode": "train", "global_step": 619, "epoch": 0.062179809141135106, "loss": 0.1221, "grad_norm": 28.908626556396484, "learning_rate": 8.127272727272728e-06, "num_tokens": 1095532.0, "completions/mean_length": 28.875, "completions/min_length": 24.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.875, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.45595115423202515, "rewards/meter/std": 0.3782821595668793, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9545157551765442, "rewards/repeat_soft/std": 0.09621007740497589, "rewards/judge_quality/mean": 0.8575000166893005, "rewards/judge_quality/std": 0.176776722073555, "rewards/total_composite/mean": 0.5886853337287903, "rewards/total_composite/std": 0.21713681519031525, "reward": 0.5886853337287903, "reward_std": 0.21713680028915405, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13166476786136627, "sampling/sampling_logp_difference/max": 1.7379283905029297, "sampling/importance_sampling_ratio/min": 0.17588438093662262, "sampling/importance_sampling_ratio/mean": 0.9892944097518921, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4909357912838459, "clip_ratio/low_mean": 0.07224462367594242, "clip_ratio/low_min": 0.07224462367594242, "clip_ratio/high_mean": 0.041193182580173016, "clip_ratio/high_max": 0.041193182580173016, "clip_ratio/region_mean": 0.11343780625611544, "reward_total_mean": 0.5886853337287903, "reward_meter_mean": 0.45595115423202515, "reward_meter_std": 0.3782821595668793, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9545157551765442, "reward_repeat_soft_std": 0.09621007740497589, "reward_judge_quality_mean": 0.8575000166893005, "reward_judge_quality_std": 0.176776722073555, "reward_total_composite_mean": 0.5886853337287903, "reward_total_composite_std": 0.21713681519031525} {"timestamp_utc": "2026-04-13T08:55:11Z", "mode": "train", "global_step": 620, "epoch": 0.0622802611752888, "loss": 0.0685, "grad_norm": 12.25234317779541, "learning_rate": 8.124242424242424e-06, "num_tokens": 1097493.0, "completions/mean_length": 57.125, "completions/min_length": 51.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.125, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.5478992462158203, "rewards/meter/std": 0.46492108702659607, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6984058618545532, "rewards/repeat_soft/std": 0.09894108027219772, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.434698224067688, "rewards/total_composite/std": 0.1239008679986, "reward": 0.434698224067688, "reward_std": 0.1239008754491806, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11980719119310379, "sampling/sampling_logp_difference/max": 1.217726707458496, "sampling/importance_sampling_ratio/min": 0.2959020733833313, "sampling/importance_sampling_ratio/mean": 1.0205646753311157, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0054671689867973, "clip_ratio/low_mean": 0.05324944853782654, "clip_ratio/low_min": 0.05324944853782654, "clip_ratio/high_mean": 0.09571819379925728, "clip_ratio/high_max": 0.09571819379925728, "clip_ratio/region_mean": 0.14896764233708382, "reward_total_mean": 0.434698224067688, "reward_meter_mean": 0.5478992462158203, "reward_meter_std": 0.46492108702659607, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6984058618545532, "reward_repeat_soft_std": 0.09894108027219772, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.434698224067688, "reward_total_composite_std": 0.1239008679986} {"timestamp_utc": "2026-04-13T08:55:17Z", "mode": "train", "global_step": 621, "epoch": 0.06238071320944249, "loss": 0.0479, "grad_norm": 17.410306930541992, "learning_rate": 8.121212121212121e-06, "num_tokens": 1099127.0, "completions/mean_length": 35.25, "completions/min_length": 33.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.25, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.3701145648956299, "rewards/meter/std": 0.4021562933921814, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.999520480632782, "rewards/repeat_soft/std": 0.0008879249216988683, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.45440980792045593, "rewards/total_composite/std": 0.10795284807682037, "reward": 0.45440980792045593, "reward_std": 0.10795284062623978, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18046563863754272, "sampling/sampling_logp_difference/max": 2.2352566719055176, "sampling/importance_sampling_ratio/min": 0.10696467012166977, "sampling/importance_sampling_ratio/mean": 1.031589388847351, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9421701580286026, "clip_ratio/low_mean": 0.09570794692263007, "clip_ratio/low_min": 0.09570794692263007, "clip_ratio/high_mean": 0.05011655203998089, "clip_ratio/high_max": 0.05011655203998089, "clip_ratio/region_mean": 0.14582449896261096, "reward_total_mean": 0.45440980792045593, "reward_meter_mean": 0.3701145648956299, "reward_meter_std": 0.4021562933921814, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.999520480632782, "reward_repeat_soft_std": 0.0008879249216988683, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.45440980792045593, "reward_total_composite_std": 0.10795284807682037} {"timestamp_utc": "2026-04-13T08:55:28Z", "mode": "train", "global_step": 622, "epoch": 0.06248116524359618, "loss": -0.1313, "grad_norm": 3.2621078491210938, "learning_rate": 8.118181818181819e-06, "num_tokens": 1100870.0, "completions/mean_length": 115.875, "completions/min_length": 54.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 59.28571701049805, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.6296329498291016, "rewards/meter/std": 0.3438984155654907, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8968924283981323, "rewards/repeat_soft/std": 0.08101747930049896, "rewards/judge_quality/mean": 0.4112499952316284, "rewards/judge_quality/std": 0.1797965168952942, "rewards/total_composite/mean": 0.44532322883605957, "rewards/total_composite/std": 0.20290635526180267, "reward": 0.44532322883605957, "reward_std": 0.20290635526180267, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14242790639400482, "sampling/sampling_logp_difference/max": 1.327103614807129, "sampling/importance_sampling_ratio/min": 0.26524439454078674, "sampling/importance_sampling_ratio/mean": 1.0022187232971191, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9111338779330254, "clip_ratio/low_mean": 0.019625603687018156, "clip_ratio/low_min": 0.019625603687018156, "clip_ratio/high_mean": 0.08964197617024183, "clip_ratio/high_max": 0.08964197617024183, "clip_ratio/region_mean": 0.10926757985725999, "reward_total_mean": 0.44532322883605957, "reward_meter_mean": 0.6296329498291016, "reward_meter_std": 0.3438984155654907, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8968924283981323, "reward_repeat_soft_std": 0.08101747930049896, "reward_judge_quality_mean": 0.4112499952316284, "reward_judge_quality_std": 0.1797965168952942, "reward_total_composite_mean": 0.44532322883605957, "reward_total_composite_std": 0.20290635526180267} {"timestamp_utc": "2026-04-13T08:55:34Z", "mode": "train", "global_step": 623, "epoch": 0.06258161727774987, "loss": -0.0309, "grad_norm": 17.54698371887207, "learning_rate": 8.115151515151516e-06, "num_tokens": 1102515.0, "completions/mean_length": 31.625, "completions/min_length": 27.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.625, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.9827060699462891, "rewards/meter/std": 0.012876130640506744, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.958416223526001, "rewards/repeat_soft/std": 0.03597824275493622, "rewards/judge_quality/mean": 0.5487500429153442, "rewards/judge_quality/std": 0.22937415540218353, "rewards/total_composite/mean": 0.6940488815307617, "rewards/total_composite/std": 0.14654476940631866, "reward": 0.6940488815307617, "reward_std": 0.14654478430747986, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14996962249279022, "sampling/sampling_logp_difference/max": 1.2179861068725586, "sampling/importance_sampling_ratio/min": 0.295825332403183, "sampling/importance_sampling_ratio/mean": 1.0346026420593262, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0386882349848747, "clip_ratio/low_mean": 0.11467740964144468, "clip_ratio/low_min": 0.11467740964144468, "clip_ratio/high_mean": 0.043174341320991516, "clip_ratio/high_max": 0.043174341320991516, "clip_ratio/region_mean": 0.1578517509624362, "reward_total_mean": 0.6940488815307617, "reward_meter_mean": 0.9827060699462891, "reward_meter_std": 0.012876130640506744, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.958416223526001, "reward_repeat_soft_std": 0.03597824275493622, "reward_judge_quality_mean": 0.5487500429153442, "reward_judge_quality_std": 0.22937415540218353, "reward_total_composite_mean": 0.6940488815307617, "reward_total_composite_std": 0.14654476940631866} {"timestamp_utc": "2026-04-13T08:55:40Z", "mode": "train", "global_step": 624, "epoch": 0.06268206931190357, "loss": 0.132, "grad_norm": 30.478708267211914, "learning_rate": 8.112121212121213e-06, "num_tokens": 1103830.0, "completions/mean_length": 17.375, "completions/min_length": 15.0, "completions/max_length": 21.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 17.375, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 21.0, "rewards/meter/mean": 0.6895481944084167, "rewards/meter/std": 0.3371638059616089, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.5350354313850403, "rewards/total_composite/std": 0.09472057223320007, "reward": 0.5350354313850403, "reward_std": 0.09472055733203888, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20352110266685486, "sampling/sampling_logp_difference/max": 1.9299743175506592, "sampling/importance_sampling_ratio/min": 0.14515192806720734, "sampling/importance_sampling_ratio/mean": 0.9885327816009521, "sampling/importance_sampling_ratio/max": 1.9450926780700684, "entropy": 1.3484907150268555, "clip_ratio/low_mean": 0.08521303255110979, "clip_ratio/low_min": 0.08521303255110979, "clip_ratio/high_mean": 0.09353554248809814, "clip_ratio/high_max": 0.09353554248809814, "clip_ratio/region_mean": 0.17874857503920794, "reward_total_mean": 0.5350354313850403, "reward_meter_mean": 0.6895481944084167, "reward_meter_std": 0.3371638059616089, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.5350354313850403, "reward_total_composite_std": 0.09472057223320007} {"timestamp_utc": "2026-04-13T08:55:47Z", "mode": "train", "global_step": 625, "epoch": 0.06278252134605726, "loss": 0.0167, "grad_norm": 11.994166374206543, "learning_rate": 8.10909090909091e-06, "num_tokens": 1105301.0, "completions/mean_length": 30.875, "completions/min_length": 29.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.875, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.8243540525436401, "rewards/meter/std": 0.3199911117553711, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9536811113357544, "rewards/repeat_soft/std": 0.05209203436970711, "rewards/judge_quality/mean": 0.5687500238418579, "rewards/judge_quality/std": 0.19111984968185425, "rewards/total_composite/mean": 0.6183902025222778, "rewards/total_composite/std": 0.10725264996290207, "reward": 0.6183902025222778, "reward_std": 0.10725263506174088, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1609441339969635, "sampling/sampling_logp_difference/max": 1.681645393371582, "sampling/importance_sampling_ratio/min": 0.18606756627559662, "sampling/importance_sampling_ratio/mean": 0.9962921738624573, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0047850385308266, "clip_ratio/low_mean": 0.07782866479828954, "clip_ratio/low_min": 0.07782866479828954, "clip_ratio/high_mean": 0.09491125121712685, "clip_ratio/high_max": 0.09491125121712685, "clip_ratio/region_mean": 0.17273991601541638, "reward_total_mean": 0.6183902025222778, "reward_meter_mean": 0.8243540525436401, "reward_meter_std": 0.3199911117553711, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9536811113357544, "reward_repeat_soft_std": 0.05209203436970711, "reward_judge_quality_mean": 0.5687500238418579, "reward_judge_quality_std": 0.19111984968185425, "reward_total_composite_mean": 0.6183902025222778, "reward_total_composite_std": 0.10725264996290207} {"timestamp_utc": "2026-04-13T08:55:53Z", "mode": "train", "global_step": 626, "epoch": 0.06288297338021095, "loss": 0.0544, "grad_norm": 20.4157772064209, "learning_rate": 8.106060606060606e-06, "num_tokens": 1106803.0, "completions/mean_length": 35.75, "completions/min_length": 30.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.75, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.7206422090530396, "rewards/meter/std": 0.34192830324172974, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9348602890968323, "rewards/repeat_soft/std": 0.10164881497621536, "rewards/judge_quality/mean": 0.5850000381469727, "rewards/judge_quality/std": 0.15297061204910278, "rewards/total_composite/mean": 0.6152657270431519, "rewards/total_composite/std": 0.1582135111093521, "reward": 0.6152657270431519, "reward_std": 0.15821349620819092, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18463172018527985, "sampling/sampling_logp_difference/max": 1.4099512100219727, "sampling/importance_sampling_ratio/min": 0.24415519833564758, "sampling/importance_sampling_ratio/mean": 1.0475335121154785, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6291310265660286, "clip_ratio/low_mean": 0.1022753482684493, "clip_ratio/low_min": 0.1022753482684493, "clip_ratio/high_mean": 0.09265873208642006, "clip_ratio/high_max": 0.09265873208642006, "clip_ratio/region_mean": 0.19493408035486937, "reward_total_mean": 0.6152657270431519, "reward_meter_mean": 0.7206422090530396, "reward_meter_std": 0.34192830324172974, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9348602890968323, "reward_repeat_soft_std": 0.10164881497621536, "reward_judge_quality_mean": 0.5850000381469727, "reward_judge_quality_std": 0.15297061204910278, "reward_total_composite_mean": 0.6152657270431519, "reward_total_composite_std": 0.1582135111093521} {"timestamp_utc": "2026-04-13T08:55:59Z", "mode": "train", "global_step": 627, "epoch": 0.06298342541436464, "loss": 0.0083, "grad_norm": 12.007946968078613, "learning_rate": 8.103030303030303e-06, "num_tokens": 1108759.0, "completions/mean_length": 74.5, "completions/min_length": 63.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.5, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.8114852905273438, "rewards/meter/std": 0.2701057493686676, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.868984580039978, "rewards/repeat_soft/std": 0.06701181828975677, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.5804606676101685, "rewards/total_composite/std": 0.10382802784442902, "reward": 0.5804606676101685, "reward_std": 0.10382802039384842, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1396416574716568, "sampling/sampling_logp_difference/max": 1.781585693359375, "sampling/importance_sampling_ratio/min": 0.16837096214294434, "sampling/importance_sampling_ratio/mean": 1.0330348014831543, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.169243462383747, "clip_ratio/low_mean": 0.028571429662406445, "clip_ratio/low_min": 0.028571429662406445, "clip_ratio/high_mean": 0.09981811698526144, "clip_ratio/high_max": 0.09981811698526144, "clip_ratio/region_mean": 0.12838954664766788, "reward_total_mean": 0.5804606676101685, "reward_meter_mean": 0.8114852905273438, "reward_meter_std": 0.2701057493686676, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.868984580039978, "reward_repeat_soft_std": 0.06701181828975677, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.5804606676101685, "reward_total_composite_std": 0.10382802784442902} {"timestamp_utc": "2026-04-13T08:56:11Z", "mode": "train", "global_step": 628, "epoch": 0.06308387744851833, "loss": -0.193, "grad_norm": 2.5380585193634033, "learning_rate": 8.1e-06, "num_tokens": 1110957.0, "completions/mean_length": 156.75, "completions/min_length": 99.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 106.00000762939453, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 120.0, "rewards/meter/mean": 0.8502894639968872, "rewards/meter/std": 0.2771815061569214, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.5566608309745789, "rewards/repeat_soft/std": 0.11484144628047943, "rewards/judge_quality/mean": 0.2212499976158142, "rewards/judge_quality/std": 0.10842212289571762, "rewards/total_composite/mean": 0.36367928981781006, "rewards/total_composite/std": 0.1626027375459671, "reward": 0.36367928981781006, "reward_std": 0.1626027375459671, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08519712090492249, "sampling/sampling_logp_difference/max": 1.9697628021240234, "sampling/importance_sampling_ratio/min": 0.13948993384838104, "sampling/importance_sampling_ratio/mean": 1.0176167488098145, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6992944739758968, "clip_ratio/low_mean": 0.0071428571827709675, "clip_ratio/low_min": 0.0071428571827709675, "clip_ratio/high_mean": 0.051890133414417505, "clip_ratio/high_max": 0.051890133414417505, "clip_ratio/region_mean": 0.05903299059718847, "reward_total_mean": 0.36367928981781006, "reward_meter_mean": 0.8502894639968872, "reward_meter_std": 0.2771815061569214, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.5566608309745789, "reward_repeat_soft_std": 0.11484144628047943, "reward_judge_quality_mean": 0.2212499976158142, "reward_judge_quality_std": 0.10842212289571762, "reward_total_composite_mean": 0.36367928981781006, "reward_total_composite_std": 0.1626027375459671} {"timestamp_utc": "2026-04-13T08:56:17Z", "mode": "train", "global_step": 629, "epoch": 0.06318432948267202, "loss": 0.0398, "grad_norm": 18.1601505279541, "learning_rate": 8.096969696969698e-06, "num_tokens": 1112426.0, "completions/mean_length": 30.625, "completions/min_length": 29.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.625, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.5472124814987183, "rewards/meter/std": 0.3712937533855438, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9427904486656189, "rewards/repeat_soft/std": 0.07211591303348541, "rewards/judge_quality/mean": 0.4024999737739563, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.4785916209220886, "rewards/total_composite/std": 0.09244726598262787, "reward": 0.4785916209220886, "reward_std": 0.09244726598262787, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18447519838809967, "sampling/sampling_logp_difference/max": 1.5100622177124023, "sampling/importance_sampling_ratio/min": 0.22089622914791107, "sampling/importance_sampling_ratio/mean": 1.0286273956298828, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5972175747156143, "clip_ratio/low_mean": 0.05895900260657072, "clip_ratio/low_min": 0.05895900260657072, "clip_ratio/high_mean": 0.10901000909507275, "clip_ratio/high_max": 0.10901000909507275, "clip_ratio/region_mean": 0.16796901170164347, "reward_total_mean": 0.4785916209220886, "reward_meter_mean": 0.5472124814987183, "reward_meter_std": 0.3712937533855438, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9427904486656189, "reward_repeat_soft_std": 0.07211591303348541, "reward_judge_quality_mean": 0.4024999737739563, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.4785916209220886, "reward_total_composite_std": 0.09244726598262787} {"timestamp_utc": "2026-04-13T08:56:28Z", "mode": "train", "global_step": 630, "epoch": 0.06328478151682572, "loss": -0.0552, "grad_norm": 1.3623456954956055, "learning_rate": 8.093939393939395e-06, "num_tokens": 1114077.0, "completions/mean_length": 398.375, "completions/min_length": 57.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.75, "completions/mean_terminated_length": 57.5, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.34584999084472656, "rewards/meter/std": 0.3909425735473633, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.37533053755760193, "rewards/hard_gate/mean": 0.25, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.987758457660675, "rewards/repeat_soft/std": 0.016199786216020584, "rewards/judge_quality/mean": 0.22625000774860382, "rewards/judge_quality/std": 0.3057047426700592, "rewards/total_composite/mean": 0.17963233590126038, "rewards/total_composite/std": 0.33561691641807556, "reward": 0.17963233590126038, "reward_std": 0.33561691641807556, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21470822393894196, "sampling/sampling_logp_difference/max": 1.2154841423034668, "sampling/importance_sampling_ratio/min": 0.2965663969516754, "sampling/importance_sampling_ratio/mean": 1.0532341003417969, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.37769442796707153, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.045674530789256096, "clip_ratio/high_max": 0.045674530789256096, "clip_ratio/region_mean": 0.045674530789256096, "reward_total_mean": 0.17963233590126038, "reward_meter_mean": 0.34584999084472656, "reward_meter_std": 0.3909425735473633, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.37533053755760193, "reward_hard_gate_mean": 0.25, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.987758457660675, "reward_repeat_soft_std": 0.016199786216020584, "reward_judge_quality_mean": 0.22625000774860382, "reward_judge_quality_std": 0.3057047426700592, "reward_total_composite_mean": 0.17963233590126038, "reward_total_composite_std": 0.33561691641807556} {"timestamp_utc": "2026-04-13T08:56:35Z", "mode": "train", "global_step": 631, "epoch": 0.06338523355097941, "loss": 0.019, "grad_norm": 16.864667892456055, "learning_rate": 8.090909090909092e-06, "num_tokens": 1115731.0, "completions/mean_length": 49.75, "completions/min_length": 36.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.75, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.3106288015842438, "rewards/meter/std": 0.25897109508514404, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9419691562652588, "rewards/repeat_soft/std": 0.05332150682806969, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.4362753629684448, "rewards/total_composite/std": 0.0686321035027504, "reward": 0.4362753629684448, "reward_std": 0.0686321035027504, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19420987367630005, "sampling/sampling_logp_difference/max": 2.466726779937744, "sampling/importance_sampling_ratio/min": 0.08486217260360718, "sampling/importance_sampling_ratio/mean": 1.00111722946167, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8492881581187248, "clip_ratio/low_mean": 0.11858218722045422, "clip_ratio/low_min": 0.11858218722045422, "clip_ratio/high_mean": 0.0524417320266366, "clip_ratio/high_max": 0.0524417320266366, "clip_ratio/region_mean": 0.17102391924709082, "reward_total_mean": 0.4362753629684448, "reward_meter_mean": 0.3106288015842438, "reward_meter_std": 0.25897109508514404, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9419691562652588, "reward_repeat_soft_std": 0.05332150682806969, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.4362753629684448, "reward_total_composite_std": 0.0686321035027504} {"timestamp_utc": "2026-04-13T08:56:46Z", "mode": "train", "global_step": 632, "epoch": 0.0634856855851331, "loss": -0.0898, "grad_norm": 3.517648458480835, "learning_rate": 8.08787878787879e-06, "num_tokens": 1117377.0, "completions/mean_length": 99.75, "completions/min_length": 34.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 40.85714340209961, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.8064794540405273, "rewards/meter/std": 0.3122844099998474, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9868118762969971, "rewards/repeat_soft/std": 0.025085289031267166, "rewards/judge_quality/mean": 0.51500004529953, "rewards/judge_quality/std": 0.2660827040672302, "rewards/total_composite/mean": 0.5743173360824585, "rewards/total_composite/std": 0.2833027243614197, "reward": 0.5743173360824585, "reward_std": 0.2833027243614197, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1882508546113968, "sampling/sampling_logp_difference/max": 1.2981586456298828, "sampling/importance_sampling_ratio/min": 0.27303409576416016, "sampling/importance_sampling_ratio/mean": 1.0191739797592163, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3905847817659378, "clip_ratio/low_mean": 0.04161173291504383, "clip_ratio/low_min": 0.04161173291504383, "clip_ratio/high_mean": 0.10609741974622011, "clip_ratio/high_max": 0.10609741974622011, "clip_ratio/region_mean": 0.14770915266126394, "reward_total_mean": 0.5743173360824585, "reward_meter_mean": 0.8064794540405273, "reward_meter_std": 0.3122844099998474, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9868118762969971, "reward_repeat_soft_std": 0.025085289031267166, "reward_judge_quality_mean": 0.51500004529953, "reward_judge_quality_std": 0.2660827040672302, "reward_total_composite_mean": 0.5743173360824585, "reward_total_composite_std": 0.2833027243614197} {"timestamp_utc": "2026-04-13T08:56:53Z", "mode": "train", "global_step": 633, "epoch": 0.06358613761928679, "loss": 0.0163, "grad_norm": 8.809235572814941, "learning_rate": 8.084848484848485e-06, "num_tokens": 1119738.0, "completions/mean_length": 93.125, "completions/min_length": 69.0, "completions/max_length": 131.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.125, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.9915623664855957, "rewards/meter/std": 0.004384336993098259, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6404680013656616, "rewards/repeat_soft/std": 0.13526977598667145, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.10966669768095016, "rewards/total_composite/mean": 0.5077430009841919, "rewards/total_composite/std": 0.079136922955513, "reward": 0.5077430009841919, "reward_std": 0.079136922955513, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14437268674373627, "sampling/sampling_logp_difference/max": 2.0607922077178955, "sampling/importance_sampling_ratio/min": 0.12735304236412048, "sampling/importance_sampling_ratio/mean": 1.0201661586761475, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1471482813358307, "clip_ratio/low_mean": 0.05252622161060572, "clip_ratio/low_min": 0.05252622161060572, "clip_ratio/high_mean": 0.08306257333606482, "clip_ratio/high_max": 0.08306257333606482, "clip_ratio/region_mean": 0.13558879494667053, "reward_total_mean": 0.5077430009841919, "reward_meter_mean": 0.9915623664855957, "reward_meter_std": 0.004384336993098259, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6404680013656616, "reward_repeat_soft_std": 0.13526977598667145, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.10966669768095016, "reward_total_composite_mean": 0.5077430009841919, "reward_total_composite_std": 0.079136922955513} {"timestamp_utc": "2026-04-13T08:57:00Z", "mode": "train", "global_step": 634, "epoch": 0.06368658965344048, "loss": -0.088, "grad_norm": 17.352680206298828, "learning_rate": 8.081818181818182e-06, "num_tokens": 1121285.0, "completions/mean_length": 42.375, "completions/min_length": 29.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.375, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.4564714729785919, "rewards/meter/std": 0.3743678331375122, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9164788126945496, "rewards/repeat_soft/std": 0.06086752563714981, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.4197845458984375, "rewards/total_composite/std": 0.19420462846755981, "reward": 0.4197845458984375, "reward_std": 0.19420461356639862, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1582125574350357, "sampling/sampling_logp_difference/max": 2.0730929374694824, "sampling/importance_sampling_ratio/min": 0.1257961094379425, "sampling/importance_sampling_ratio/mean": 1.0250484943389893, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0219487696886063, "clip_ratio/low_mean": 0.06842627469450235, "clip_ratio/low_min": 0.06842627469450235, "clip_ratio/high_mean": 0.08998904563486576, "clip_ratio/high_max": 0.08998904563486576, "clip_ratio/region_mean": 0.15841532032936811, "reward_total_mean": 0.4197845458984375, "reward_meter_mean": 0.4564714729785919, "reward_meter_std": 0.3743678331375122, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9164788126945496, "reward_repeat_soft_std": 0.06086752563714981, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.4197845458984375, "reward_total_composite_std": 0.19420462846755981} {"timestamp_utc": "2026-04-13T08:57:06Z", "mode": "train", "global_step": 635, "epoch": 0.06378704168759418, "loss": -0.0257, "grad_norm": 13.63864517211914, "learning_rate": 8.07878787878788e-06, "num_tokens": 1122979.0, "completions/mean_length": 51.75, "completions/min_length": 37.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.75, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.6257561445236206, "rewards/meter/std": 0.3036348819732666, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7920382022857666, "rewards/repeat_soft/std": 0.0841648131608963, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.5136308073997498, "rewards/total_composite/std": 0.09853413701057434, "reward": 0.5136308073997498, "reward_std": 0.09853413701057434, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14734230935573578, "sampling/sampling_logp_difference/max": 2.8847289085388184, "sampling/importance_sampling_ratio/min": 0.0558699332177639, "sampling/importance_sampling_ratio/mean": 1.001219630241394, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6683292649686337, "clip_ratio/low_mean": 0.024950592778623104, "clip_ratio/low_min": 0.024950592778623104, "clip_ratio/high_mean": 0.09042746387422085, "clip_ratio/high_max": 0.09042746387422085, "clip_ratio/region_mean": 0.11537805665284395, "reward_total_mean": 0.5136308073997498, "reward_meter_mean": 0.6257561445236206, "reward_meter_std": 0.3036348819732666, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7920382022857666, "reward_repeat_soft_std": 0.0841648131608963, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.5136308073997498, "reward_total_composite_std": 0.09853413701057434} {"timestamp_utc": "2026-04-13T08:57:17Z", "mode": "train", "global_step": 636, "epoch": 0.06388749372174786, "loss": -0.1102, "grad_norm": 5.966771125793457, "learning_rate": 8.075757575757577e-06, "num_tokens": 1124740.0, "completions/mean_length": 109.125, "completions/min_length": 36.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 51.57143020629883, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.1405440866947174, "rewards/meter/std": 0.17522665858268738, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9446536302566528, "rewards/repeat_soft/std": 0.06525295972824097, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.13593590259552002, "rewards/total_composite/mean": 0.27637478709220886, "rewards/total_composite/std": 0.17110449075698853, "reward": 0.27637478709220886, "reward_std": 0.17110447585582733, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.27482447028160095, "sampling/sampling_logp_difference/max": 3.4003939628601074, "sampling/importance_sampling_ratio/min": 0.03336012363433838, "sampling/importance_sampling_ratio/mean": 0.9913470149040222, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8371185436844826, "clip_ratio/low_mean": 0.03125, "clip_ratio/low_min": 0.03125, "clip_ratio/high_mean": 0.157029053196311, "clip_ratio/high_max": 0.157029053196311, "clip_ratio/region_mean": 0.188279053196311, "reward_total_mean": 0.27637478709220886, "reward_meter_mean": 0.1405440866947174, "reward_meter_std": 0.17522665858268738, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9446536302566528, "reward_repeat_soft_std": 0.06525295972824097, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.13593590259552002, "reward_total_composite_mean": 0.27637478709220886, "reward_total_composite_std": 0.17110449075698853} {"timestamp_utc": "2026-04-13T08:57:24Z", "mode": "train", "global_step": 637, "epoch": 0.06398794575590155, "loss": 0.0412, "grad_norm": 15.5023832321167, "learning_rate": 8.072727272727274e-06, "num_tokens": 1126466.0, "completions/mean_length": 40.75, "completions/min_length": 31.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.75, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.5743567943572998, "rewards/meter/std": 0.36496245861053467, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9650839567184448, "rewards/repeat_soft/std": 0.0464431531727314, "rewards/judge_quality/mean": 0.668749988079071, "rewards/judge_quality/std": 0.25842589139938354, "rewards/total_composite/mean": 0.594261884689331, "rewards/total_composite/std": 0.17329005897045135, "reward": 0.594261884689331, "reward_std": 0.17329005897045135, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15703284740447998, "sampling/sampling_logp_difference/max": 1.3353967666625977, "sampling/importance_sampling_ratio/min": 0.2630538046360016, "sampling/importance_sampling_ratio/mean": 0.9977946877479553, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.869310550391674, "clip_ratio/low_mean": 0.046516415663063526, "clip_ratio/low_min": 0.046516415663063526, "clip_ratio/high_mean": 0.0637398287653923, "clip_ratio/high_max": 0.0637398287653923, "clip_ratio/region_mean": 0.11025624442845583, "reward_total_mean": 0.594261884689331, "reward_meter_mean": 0.5743567943572998, "reward_meter_std": 0.36496245861053467, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9650839567184448, "reward_repeat_soft_std": 0.0464431531727314, "reward_judge_quality_mean": 0.668749988079071, "reward_judge_quality_std": 0.25842589139938354, "reward_total_composite_mean": 0.594261884689331, "reward_total_composite_std": 0.17329005897045135} {"timestamp_utc": "2026-04-13T08:57:30Z", "mode": "train", "global_step": 638, "epoch": 0.06408839779005525, "loss": -0.0425, "grad_norm": 12.020691871643066, "learning_rate": 8.069696969696971e-06, "num_tokens": 1128476.0, "completions/mean_length": 58.25, "completions/min_length": 46.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.25, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.5777385234832764, "rewards/meter/std": 0.3327678442001343, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7816901206970215, "rewards/repeat_soft/std": 0.17234176397323608, "rewards/judge_quality/mean": 0.4737499952316284, "rewards/judge_quality/std": 0.16291432082653046, "rewards/total_composite/mean": 0.483951210975647, "rewards/total_composite/std": 0.14631693065166473, "reward": 0.483951210975647, "reward_std": 0.14631694555282593, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15901510417461395, "sampling/sampling_logp_difference/max": 1.4788970947265625, "sampling/importance_sampling_ratio/min": 0.22788889706134796, "sampling/importance_sampling_ratio/mean": 1.0201103687286377, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9441844969987869, "clip_ratio/low_mean": 0.07208479847759008, "clip_ratio/low_min": 0.07208479847759008, "clip_ratio/high_mean": 0.07776338700205088, "clip_ratio/high_max": 0.07776338700205088, "clip_ratio/region_mean": 0.14984818547964096, "reward_total_mean": 0.483951210975647, "reward_meter_mean": 0.5777385234832764, "reward_meter_std": 0.3327678442001343, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7816901206970215, "reward_repeat_soft_std": 0.17234176397323608, "reward_judge_quality_mean": 0.4737499952316284, "reward_judge_quality_std": 0.16291432082653046, "reward_total_composite_mean": 0.483951210975647, "reward_total_composite_std": 0.14631693065166473} {"timestamp_utc": "2026-04-13T08:57:41Z", "mode": "train", "global_step": 639, "epoch": 0.06418884982420894, "loss": -0.1101, "grad_norm": 3.4394798278808594, "learning_rate": 8.066666666666667e-06, "num_tokens": 1130121.0, "completions/mean_length": 97.625, "completions/min_length": 36.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 38.42857360839844, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.9434645771980286, "rewards/meter/std": 0.0684240534901619, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9191547632217407, "rewards/repeat_soft/std": 0.07343998551368713, "rewards/judge_quality/mean": 0.4737499952316284, "rewards/judge_quality/std": 0.2546110451221466, "rewards/total_composite/mean": 0.580635666847229, "rewards/total_composite/std": 0.2574915587902069, "reward": 0.580635666847229, "reward_std": 0.2574915587902069, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1892760843038559, "sampling/sampling_logp_difference/max": 1.8836870193481445, "sampling/importance_sampling_ratio/min": 0.15202854573726654, "sampling/importance_sampling_ratio/mean": 1.0117923021316528, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4847816228866577, "clip_ratio/low_mean": 0.0243055559694767, "clip_ratio/low_min": 0.0243055559694767, "clip_ratio/high_mean": 0.1251265201717615, "clip_ratio/high_max": 0.1251265201717615, "clip_ratio/region_mean": 0.1494320761412382, "reward_total_mean": 0.580635666847229, "reward_meter_mean": 0.9434645771980286, "reward_meter_std": 0.0684240534901619, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9191547632217407, "reward_repeat_soft_std": 0.07343998551368713, "reward_judge_quality_mean": 0.4737499952316284, "reward_judge_quality_std": 0.2546110451221466, "reward_total_composite_mean": 0.580635666847229, "reward_total_composite_std": 0.2574915587902069} {"timestamp_utc": "2026-04-13T08:57:48Z", "mode": "train", "global_step": 640, "epoch": 0.06428930185836264, "loss": 0.0052, "grad_norm": 9.830313682556152, "learning_rate": 8.063636363636364e-06, "num_tokens": 1132487.0, "completions/mean_length": 92.75, "completions/min_length": 83.0, "completions/max_length": 100.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.75, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 100.0, "rewards/meter/mean": 0.8226418495178223, "rewards/meter/std": 0.21657024323940277, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8834623098373413, "rewards/repeat_soft/std": 0.09841935336589813, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.5209257006645203, "rewards/total_composite/std": 0.23504388332366943, "reward": 0.5209257006645203, "reward_std": 0.23504388332366943, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16420626640319824, "sampling/sampling_logp_difference/max": 1.6748542785644531, "sampling/importance_sampling_ratio/min": 0.18733547627925873, "sampling/importance_sampling_ratio/mean": 1.0051250457763672, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0939573794603348, "clip_ratio/low_mean": 0.05977832060307264, "clip_ratio/low_min": 0.05977832060307264, "clip_ratio/high_mean": 0.117831876501441, "clip_ratio/high_max": 0.117831876501441, "clip_ratio/region_mean": 0.17761019710451365, "reward_total_mean": 0.5209257006645203, "reward_meter_mean": 0.8226418495178223, "reward_meter_std": 0.21657024323940277, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8834623098373413, "reward_repeat_soft_std": 0.09841935336589813, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.5209257006645203, "reward_total_composite_std": 0.23504388332366943} {"timestamp_utc": "2026-04-13T08:57:54Z", "mode": "train", "global_step": 641, "epoch": 0.06438975389251632, "loss": -0.0035, "grad_norm": 15.511313438415527, "learning_rate": 8.060606060606061e-06, "num_tokens": 1134063.0, "completions/mean_length": 26.0, "completions/min_length": 23.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.0, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.7776494026184082, "rewards/meter/std": 0.30838698148727417, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9664133191108704, "rewards/repeat_soft/std": 0.046133968979120255, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.6498969197273254, "rewards/total_composite/std": 0.16817516088485718, "reward": 0.6498969197273254, "reward_std": 0.16817517578601837, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10660341382026672, "sampling/sampling_logp_difference/max": 1.3756260871887207, "sampling/importance_sampling_ratio/min": 0.25268134474754333, "sampling/importance_sampling_ratio/mean": 1.0160659551620483, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.819943368434906, "clip_ratio/low_mean": 0.07618798967450857, "clip_ratio/low_min": 0.07618798967450857, "clip_ratio/high_mean": 0.02855603490024805, "clip_ratio/high_max": 0.02855603490024805, "clip_ratio/region_mean": 0.10474402457475662, "reward_total_mean": 0.6498969197273254, "reward_meter_mean": 0.7776494026184082, "reward_meter_std": 0.30838698148727417, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9664133191108704, "reward_repeat_soft_std": 0.046133968979120255, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.6498969197273254, "reward_total_composite_std": 0.16817516088485718} {"timestamp_utc": "2026-04-13T08:58:01Z", "mode": "train", "global_step": 642, "epoch": 0.06449020592667001, "loss": -0.0551, "grad_norm": 13.684636116027832, "learning_rate": 8.057575757575759e-06, "num_tokens": 1135927.0, "completions/mean_length": 68.0, "completions/min_length": 54.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.0, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.47336992621421814, "rewards/meter/std": 0.39060527086257935, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8227493166923523, "rewards/repeat_soft/std": 0.12701508402824402, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.35991573333740234, "rewards/total_composite/std": 0.17331789433956146, "reward": 0.35991573333740234, "reward_std": 0.17331789433956146, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1437559574842453, "sampling/sampling_logp_difference/max": 2.008169174194336, "sampling/importance_sampling_ratio/min": 0.1342342048883438, "sampling/importance_sampling_ratio/mean": 1.015454649925232, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.000231422483921, "clip_ratio/low_mean": 0.09347231686115265, "clip_ratio/low_min": 0.09347231686115265, "clip_ratio/high_mean": 0.08196821622550488, "clip_ratio/high_max": 0.08196821622550488, "clip_ratio/region_mean": 0.17544053308665752, "reward_total_mean": 0.35991573333740234, "reward_meter_mean": 0.47336992621421814, "reward_meter_std": 0.39060527086257935, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8227493166923523, "reward_repeat_soft_std": 0.12701508402824402, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.35991573333740234, "reward_total_composite_std": 0.17331789433956146} {"timestamp_utc": "2026-04-13T08:58:12Z", "mode": "train", "global_step": 643, "epoch": 0.06459065796082371, "loss": -0.0572, "grad_norm": 3.3506882190704346, "learning_rate": 8.054545454545454e-06, "num_tokens": 1137310.0, "completions/mean_length": 79.875, "completions/min_length": 16.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 18.142858505249023, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 20.0, "rewards/meter/mean": 0.48889678716659546, "rewards/meter/std": 0.41893470287323, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9566287994384766, "rewards/repeat_soft/std": 0.016606280580163002, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.13617216050624847, "rewards/total_composite/mean": 0.4388635754585266, "rewards/total_composite/std": 0.20644930005073547, "reward": 0.4388635754585266, "reward_std": 0.20644930005073547, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1394256204366684, "sampling/sampling_logp_difference/max": 1.1767903566360474, "sampling/importance_sampling_ratio/min": 0.3082665801048279, "sampling/importance_sampling_ratio/mean": 1.0233968496322632, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8707086369395256, "clip_ratio/low_mean": 0.055800653994083405, "clip_ratio/low_min": 0.055800653994083405, "clip_ratio/high_mean": 0.09032346680760384, "clip_ratio/high_max": 0.09032346680760384, "clip_ratio/region_mean": 0.14612412080168724, "reward_total_mean": 0.4388635754585266, "reward_meter_mean": 0.48889678716659546, "reward_meter_std": 0.41893470287323, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9566287994384766, "reward_repeat_soft_std": 0.016606280580163002, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.13617216050624847, "reward_total_composite_mean": 0.4388635754585266, "reward_total_composite_std": 0.20644930005073547} {"timestamp_utc": "2026-04-13T08:58:20Z", "mode": "train", "global_step": 644, "epoch": 0.0646911099949774, "loss": 0.1479, "grad_norm": 12.533270835876465, "learning_rate": 8.051515151515153e-06, "num_tokens": 1139727.0, "completions/mean_length": 114.125, "completions/min_length": 86.0, "completions/max_length": 164.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 114.125, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 164.0, "rewards/meter/mean": 0.6617348790168762, "rewards/meter/std": 0.27322185039520264, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.27368009090423584, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7496739625930786, "rewards/repeat_soft/std": 0.1311352699995041, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.48523104190826416, "rewards/total_composite/std": 0.1867832988500595, "reward": 0.48523104190826416, "reward_std": 0.1867832988500595, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1406610608100891, "sampling/sampling_logp_difference/max": 3.864229202270508, "sampling/importance_sampling_ratio/min": 0.02097908779978752, "sampling/importance_sampling_ratio/mean": 0.9994625449180603, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6350109204649925, "clip_ratio/low_mean": 0.06876762956380844, "clip_ratio/low_min": 0.06876762956380844, "clip_ratio/high_mean": 0.06263576168566942, "clip_ratio/high_max": 0.06263576168566942, "clip_ratio/region_mean": 0.13140339124947786, "reward_total_mean": 0.48523104190826416, "reward_meter_mean": 0.6617348790168762, "reward_meter_std": 0.27322185039520264, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.27368009090423584, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7496739625930786, "reward_repeat_soft_std": 0.1311352699995041, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.48523104190826416, "reward_total_composite_std": 0.1867832988500595} {"timestamp_utc": "2026-04-13T08:58:27Z", "mode": "train", "global_step": 645, "epoch": 0.06479156202913108, "loss": 0.0746, "grad_norm": 10.910260200500488, "learning_rate": 8.048484848484849e-06, "num_tokens": 1141484.0, "completions/mean_length": 53.625, "completions/min_length": 45.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.625, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.27481338381767273, "rewards/meter/std": 0.3332115411758423, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7180003523826599, "rewards/repeat_soft/std": 0.19509495794773102, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.40990525484085083, "rewards/total_composite/std": 0.16834187507629395, "reward": 0.40990525484085083, "reward_std": 0.16834186017513275, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1587858498096466, "sampling/sampling_logp_difference/max": 1.695112705230713, "sampling/importance_sampling_ratio/min": 0.18357853591442108, "sampling/importance_sampling_ratio/mean": 1.024529218673706, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0420835763216019, "clip_ratio/low_mean": 0.0652546074707061, "clip_ratio/low_min": 0.0652546074707061, "clip_ratio/high_mean": 0.07155228778719902, "clip_ratio/high_max": 0.07155228778719902, "clip_ratio/region_mean": 0.13680689525790513, "reward_total_mean": 0.40990525484085083, "reward_meter_mean": 0.27481338381767273, "reward_meter_std": 0.3332115411758423, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7180003523826599, "reward_repeat_soft_std": 0.19509495794773102, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.40990525484085083, "reward_total_composite_std": 0.16834187507629395} {"timestamp_utc": "2026-04-13T08:58:33Z", "mode": "train", "global_step": 646, "epoch": 0.06489201406328478, "loss": 0.0154, "grad_norm": 16.53147315979004, "learning_rate": 8.045454545454546e-06, "num_tokens": 1142892.0, "completions/mean_length": 31.0, "completions/min_length": 29.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.0, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.8005962371826172, "rewards/meter/std": 0.32852140069007874, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9661219716072083, "rewards/repeat_soft/std": 0.03351852297782898, "rewards/judge_quality/mean": 0.5225000381469727, "rewards/judge_quality/std": 0.24598202109336853, "rewards/total_composite/mean": 0.5804761052131653, "rewards/total_composite/std": 0.29398468136787415, "reward": 0.5804761052131653, "reward_std": 0.29398468136787415, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.131461963057518, "sampling/sampling_logp_difference/max": 1.008401870727539, "sampling/importance_sampling_ratio/min": 0.3648015260696411, "sampling/importance_sampling_ratio/mean": 1.0192826986312866, "sampling/importance_sampling_ratio/max": 1.921644687652588, "entropy": 1.0241151303052902, "clip_ratio/low_mean": 0.025862068869173527, "clip_ratio/low_min": 0.025862068869173527, "clip_ratio/high_mean": 0.0889372294768691, "clip_ratio/high_max": 0.0889372294768691, "clip_ratio/region_mean": 0.11479929834604263, "reward_total_mean": 0.5804761052131653, "reward_meter_mean": 0.8005962371826172, "reward_meter_std": 0.32852140069007874, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9661219716072083, "reward_repeat_soft_std": 0.03351852297782898, "reward_judge_quality_mean": 0.5225000381469727, "reward_judge_quality_std": 0.24598202109336853, "reward_total_composite_mean": 0.5804761052131653, "reward_total_composite_std": 0.29398468136787415} {"timestamp_utc": "2026-04-13T08:58:40Z", "mode": "train", "global_step": 647, "epoch": 0.06499246609743847, "loss": 0.023, "grad_norm": 15.462921142578125, "learning_rate": 8.042424242424243e-06, "num_tokens": 1145026.0, "completions/mean_length": 73.75, "completions/min_length": 60.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.75, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.5346683263778687, "rewards/meter/std": 0.4203801155090332, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8131506443023682, "rewards/repeat_soft/std": 0.07363694161176682, "rewards/judge_quality/mean": 0.6274999976158142, "rewards/judge_quality/std": 0.18013885617256165, "rewards/total_composite/mean": 0.517020046710968, "rewards/total_composite/std": 0.1653960943222046, "reward": 0.517020046710968, "reward_std": 0.1653960943222046, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17536382377147675, "sampling/sampling_logp_difference/max": 2.880128860473633, "sampling/importance_sampling_ratio/min": 0.056127529591321945, "sampling/importance_sampling_ratio/mean": 1.0183418989181519, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9807343557476997, "clip_ratio/low_mean": 0.07348845060914755, "clip_ratio/low_min": 0.07348845060914755, "clip_ratio/high_mean": 0.06786853261291981, "clip_ratio/high_max": 0.06786853261291981, "clip_ratio/region_mean": 0.14135698322206736, "reward_total_mean": 0.517020046710968, "reward_meter_mean": 0.5346683263778687, "reward_meter_std": 0.4203801155090332, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8131506443023682, "reward_repeat_soft_std": 0.07363694161176682, "reward_judge_quality_mean": 0.6274999976158142, "reward_judge_quality_std": 0.18013885617256165, "reward_total_composite_mean": 0.517020046710968, "reward_total_composite_std": 0.1653960943222046} {"timestamp_utc": "2026-04-13T08:58:46Z", "mode": "train", "global_step": 648, "epoch": 0.06509291813159217, "loss": 0.082, "grad_norm": 18.118453979492188, "learning_rate": 8.03939393939394e-06, "num_tokens": 1146451.0, "completions/mean_length": 35.125, "completions/min_length": 28.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.125, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.47180289030075073, "rewards/meter/std": 0.3639613389968872, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9971110224723816, "rewards/repeat_soft/std": 0.002904646098613739, "rewards/judge_quality/mean": 0.5325000286102295, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.5076788663864136, "rewards/total_composite/std": 0.11163168400526047, "reward": 0.5076788663864136, "reward_std": 0.11163167655467987, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17425420880317688, "sampling/sampling_logp_difference/max": 1.1415696144104004, "sampling/importance_sampling_ratio/min": 0.31931743025779724, "sampling/importance_sampling_ratio/mean": 1.0409749746322632, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1029235646128654, "clip_ratio/low_mean": 0.1174242440611124, "clip_ratio/low_min": 0.1174242440611124, "clip_ratio/high_mean": 0.07134091667830944, "clip_ratio/high_max": 0.07134091667830944, "clip_ratio/region_mean": 0.18876516073942184, "reward_total_mean": 0.5076788663864136, "reward_meter_mean": 0.47180289030075073, "reward_meter_std": 0.3639613389968872, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9971110224723816, "reward_repeat_soft_std": 0.002904646098613739, "reward_judge_quality_mean": 0.5325000286102295, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.5076788663864136, "reward_total_composite_std": 0.11163168400526047} {"timestamp_utc": "2026-04-13T08:58:51Z", "mode": "train", "global_step": 649, "epoch": 0.06519337016574586, "loss": 0.021, "grad_norm": 22.424516677856445, "learning_rate": 8.036363636363636e-06, "num_tokens": 1148153.0, "completions/mean_length": 33.75, "completions/min_length": 29.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.75, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.623406708240509, "rewards/meter/std": 0.38744696974754333, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.971734881401062, "rewards/repeat_soft/std": 0.04234902188181877, "rewards/judge_quality/mean": 0.6737500429153442, "rewards/judge_quality/std": 0.263435423374176, "rewards/total_composite/mean": 0.6547472476959229, "rewards/total_composite/std": 0.24986574053764343, "reward": 0.6547472476959229, "reward_std": 0.24986572563648224, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15227848291397095, "sampling/sampling_logp_difference/max": 0.9737920761108398, "sampling/importance_sampling_ratio/min": 0.3776482343673706, "sampling/importance_sampling_ratio/mean": 1.0295783281326294, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1171982809901237, "clip_ratio/low_mean": 0.10233087697997689, "clip_ratio/low_min": 0.10233087697997689, "clip_ratio/high_mean": 0.06856060773134232, "clip_ratio/high_max": 0.06856060773134232, "clip_ratio/region_mean": 0.1708914847113192, "reward_total_mean": 0.6547472476959229, "reward_meter_mean": 0.623406708240509, "reward_meter_std": 0.38744696974754333, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.971734881401062, "reward_repeat_soft_std": 0.04234902188181877, "reward_judge_quality_mean": 0.6737500429153442, "reward_judge_quality_std": 0.263435423374176, "reward_total_composite_mean": 0.6547472476959229, "reward_total_composite_std": 0.24986574053764343} {"timestamp_utc": "2026-04-13T08:58:58Z", "mode": "train", "global_step": 650, "epoch": 0.06529382219989954, "loss": 0.0298, "grad_norm": 14.326630592346191, "learning_rate": 8.033333333333335e-06, "num_tokens": 1149938.0, "completions/mean_length": 59.125, "completions/min_length": 54.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.125, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.8271393179893494, "rewards/meter/std": 0.16652563214302063, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8895382881164551, "rewards/repeat_soft/std": 0.06680367887020111, "rewards/judge_quality/mean": 0.5699999928474426, "rewards/judge_quality/std": 0.16035676002502441, "rewards/total_composite/mean": 0.5602148175239563, "rewards/total_composite/std": 0.2508557438850403, "reward": 0.5602148175239563, "reward_std": 0.2508557438850403, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19041897356510162, "sampling/sampling_logp_difference/max": 2.1767501831054688, "sampling/importance_sampling_ratio/min": 0.11340949684381485, "sampling/importance_sampling_ratio/mean": 1.004520058631897, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2831703871488571, "clip_ratio/low_mean": 0.057008782401680946, "clip_ratio/low_min": 0.057008782401680946, "clip_ratio/high_mean": 0.1350773647427559, "clip_ratio/high_max": 0.1350773647427559, "clip_ratio/region_mean": 0.19208614714443684, "reward_total_mean": 0.5602148175239563, "reward_meter_mean": 0.8271393179893494, "reward_meter_std": 0.16652563214302063, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8895382881164551, "reward_repeat_soft_std": 0.06680367887020111, "reward_judge_quality_mean": 0.5699999928474426, "reward_judge_quality_std": 0.16035676002502441, "reward_total_composite_mean": 0.5602148175239563, "reward_total_composite_std": 0.2508557438850403} {"timestamp_utc": "2026-04-13T08:59:44Z", "mode": "eval", "global_step": 650, "epoch": 0.06529382219989954, "eval_loss": NaN, "eval_runtime": 45.6379, "eval_samples_per_second": 1.753, "eval_steps_per_second": 0.219, "eval_num_tokens": 1149938.0, "eval_completions/mean_length": 65.9625, "eval_completions/min_length": 28.2, "eval_completions/max_length": 173.4, "eval_completions/clipped_ratio": 0.025, "eval_completions/mean_terminated_length": 54.728572082519534, "eval_completions/min_terminated_length": 28.2, "eval_completions/max_terminated_length": 91.8, "eval_rewards/meter/mean": 0.5899448305368423, "eval_rewards/meter/std": 0.35944695919752123, "eval_rewards/count_adherence/mean": 0.9735416650772095, "eval_rewards/count_adherence/std": 0.060814641788601874, "eval_rewards/hard_gate/mean": 0.9, "eval_rewards/hard_gate/std": 0.23400336503982544, "eval_rewards/repeat_soft/mean": 0.8852393925189972, "eval_rewards/repeat_soft/std": 0.11525812894105911, "eval_rewards/judge_quality/mean": 0.48649999797344207, "eval_rewards/judge_quality/std": 0.1396345805376768, "eval_rewards/total_composite/mean": 0.4717728167772293, "eval_rewards/total_composite/std": 0.18849601075053216, "eval_reward": 0.4717728167772293, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.0842033289372921, "eval_sampling/sampling_logp_difference/max": 0.9717438220977783, "eval_sampling/importance_sampling_ratio/min": 0.38339222967624664, "eval_sampling/importance_sampling_ratio/mean": 1.019832593202591, "eval_sampling/importance_sampling_ratio/max": 1.474166202545166, "eval_entropy": 0.9724018573760986, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.4717728167772293, "eval_reward_meter_mean": 0.5899448305368423, "eval_reward_meter_std": 0.35944695919752123, "eval_reward_count_adherence_mean": 0.9735416650772095, "eval_reward_count_adherence_std": 0.060814641788601874, "eval_reward_hard_gate_mean": 0.9, "eval_reward_hard_gate_std": 0.23400336503982544, "eval_reward_repeat_soft_mean": 0.8852393925189972, "eval_reward_repeat_soft_std": 0.11525812894105911, "eval_reward_judge_quality_mean": 0.48649999797344207, "eval_reward_judge_quality_std": 0.1396345805376768, "eval_reward_total_composite_mean": 0.4717728167772293, "eval_reward_total_composite_std": 0.18849601075053216} {"timestamp_utc": "2026-04-13T08:59:53Z", "mode": "train", "global_step": 651, "epoch": 0.06539427423405324, "loss": 0.0216, "grad_norm": 12.091530799865723, "learning_rate": 8.03030303030303e-06, "num_tokens": 1151871.0, "completions/mean_length": 48.625, "completions/min_length": 41.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.625, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.981431245803833, "rewards/meter/std": 0.007213943172246218, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9497766494750977, "rewards/repeat_soft/std": 0.03229496628046036, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.610397219657898, "rewards/total_composite/std": 0.006398757919669151, "reward": 0.610397219657898, "reward_std": 0.006398768164217472, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15845441818237305, "sampling/sampling_logp_difference/max": 1.833327293395996, "sampling/importance_sampling_ratio/min": 0.15988071262836456, "sampling/importance_sampling_ratio/mean": 1.0219941139221191, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8763070181012154, "clip_ratio/low_mean": 0.03524390235543251, "clip_ratio/low_min": 0.03524390235543251, "clip_ratio/high_mean": 0.13669268880039454, "clip_ratio/high_max": 0.13669268880039454, "clip_ratio/region_mean": 0.17193659115582705, "reward_total_mean": 0.610397219657898, "reward_meter_mean": 0.981431245803833, "reward_meter_std": 0.007213943172246218, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9497766494750977, "reward_repeat_soft_std": 0.03229496628046036, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.610397219657898, "reward_total_composite_std": 0.006398757919669151} {"timestamp_utc": "2026-04-13T09:00:00Z", "mode": "train", "global_step": 652, "epoch": 0.06549472626820693, "loss": 0.0362, "grad_norm": 12.559391975402832, "learning_rate": 8.027272727272728e-06, "num_tokens": 1153579.0, "completions/mean_length": 52.5, "completions/min_length": 46.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.5, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.5216165781021118, "rewards/meter/std": 0.34197962284088135, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9827027320861816, "rewards/repeat_soft/std": 0.0072836256586015224, "rewards/judge_quality/mean": 0.7987500429153442, "rewards/judge_quality/std": 0.10091545432806015, "rewards/total_composite/mean": 0.5056251287460327, "rewards/total_composite/std": 0.24777263402938843, "reward": 0.5056251287460327, "reward_std": 0.24777261912822723, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18527501821517944, "sampling/sampling_logp_difference/max": 1.9833526611328125, "sampling/importance_sampling_ratio/min": 0.13760711252689362, "sampling/importance_sampling_ratio/mean": 1.0162345170974731, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.240620642900467, "clip_ratio/low_mean": 0.10041659697890282, "clip_ratio/low_min": 0.10041659697890282, "clip_ratio/high_mean": 0.06934077944606543, "clip_ratio/high_max": 0.06934077944606543, "clip_ratio/region_mean": 0.16975737642496824, "reward_total_mean": 0.5056251287460327, "reward_meter_mean": 0.5216165781021118, "reward_meter_std": 0.34197962284088135, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9827027320861816, "reward_repeat_soft_std": 0.0072836256586015224, "reward_judge_quality_mean": 0.7987500429153442, "reward_judge_quality_std": 0.10091545432806015, "reward_total_composite_mean": 0.5056251287460327, "reward_total_composite_std": 0.24777263402938843} {"timestamp_utc": "2026-04-13T09:00:11Z", "mode": "train", "global_step": 653, "epoch": 0.06559517830236063, "loss": -0.0844, "grad_norm": 4.386373043060303, "learning_rate": 8.024242424242425e-06, "num_tokens": 1155180.0, "completions/mean_length": 98.125, "completions/min_length": 34.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 39.0, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.7292758822441101, "rewards/meter/std": 0.42446523904800415, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9710483551025391, "rewards/repeat_soft/std": 0.023260634392499924, "rewards/judge_quality/mean": 0.6862499713897705, "rewards/judge_quality/std": 0.34221702814102173, "rewards/total_composite/mean": 0.6550273895263672, "rewards/total_composite/std": 0.3373952805995941, "reward": 0.6550273895263672, "reward_std": 0.3373952805995941, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1691875159740448, "sampling/sampling_logp_difference/max": 1.1866025924682617, "sampling/importance_sampling_ratio/min": 0.30525660514831543, "sampling/importance_sampling_ratio/mean": 1.006740927696228, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2591385692358017, "clip_ratio/low_mean": 0.05931802652776241, "clip_ratio/low_min": 0.05931802652776241, "clip_ratio/high_mean": 0.08195196930319071, "clip_ratio/high_max": 0.08195196930319071, "clip_ratio/region_mean": 0.14126999583095312, "reward_total_mean": 0.6550273895263672, "reward_meter_mean": 0.7292758822441101, "reward_meter_std": 0.42446523904800415, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9710483551025391, "reward_repeat_soft_std": 0.023260634392499924, "reward_judge_quality_mean": 0.6862499713897705, "reward_judge_quality_std": 0.34221702814102173, "reward_total_composite_mean": 0.6550273895263672, "reward_total_composite_std": 0.3373952805995941} {"timestamp_utc": "2026-04-13T09:00:18Z", "mode": "train", "global_step": 654, "epoch": 0.06569563033651432, "loss": 0.0471, "grad_norm": 14.922459602355957, "learning_rate": 8.021212121212122e-06, "num_tokens": 1157122.0, "completions/mean_length": 68.75, "completions/min_length": 49.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.75, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.7787767052650452, "rewards/meter/std": 0.23661600053310394, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8986475467681885, "rewards/repeat_soft/std": 0.11464295536279678, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.4754413068294525, "rewards/total_composite/std": 0.2196243852376938, "reward": 0.4754413068294525, "reward_std": 0.2196243703365326, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1943458765745163, "sampling/sampling_logp_difference/max": 2.324733257293701, "sampling/importance_sampling_ratio/min": 0.09780952334403992, "sampling/importance_sampling_ratio/mean": 1.0001096725463867, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5427189767360687, "clip_ratio/low_mean": 0.062207804061472416, "clip_ratio/low_min": 0.062207804061472416, "clip_ratio/high_mean": 0.11934742797166109, "clip_ratio/high_max": 0.11934742797166109, "clip_ratio/region_mean": 0.1815552320331335, "reward_total_mean": 0.4754413068294525, "reward_meter_mean": 0.7787767052650452, "reward_meter_std": 0.23661600053310394, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8986475467681885, "reward_repeat_soft_std": 0.11464295536279678, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.4754413068294525, "reward_total_composite_std": 0.2196243852376938} {"timestamp_utc": "2026-04-13T09:00:25Z", "mode": "train", "global_step": 655, "epoch": 0.065796082370668, "loss": -0.0993, "grad_norm": 13.134452819824219, "learning_rate": 8.018181818181818e-06, "num_tokens": 1158873.0, "completions/mean_length": 51.875, "completions/min_length": 40.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.875, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.22241005301475525, "rewards/meter/std": 0.26595112681388855, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8868178129196167, "rewards/repeat_soft/std": 0.061087459325790405, "rewards/judge_quality/mean": 0.5400000214576721, "rewards/judge_quality/std": 0.14957083761692047, "rewards/total_composite/mean": 0.3530135750770569, "rewards/total_composite/std": 0.17210303246974945, "reward": 0.3530135750770569, "reward_std": 0.17210301756858826, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18812473118305206, "sampling/sampling_logp_difference/max": 1.8689632415771484, "sampling/importance_sampling_ratio/min": 0.1542835384607315, "sampling/importance_sampling_ratio/mean": 1.0281426906585693, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3644793555140495, "clip_ratio/low_mean": 0.09198097139596939, "clip_ratio/low_min": 0.09198097139596939, "clip_ratio/high_mean": 0.05292119272053242, "clip_ratio/high_max": 0.05292119272053242, "clip_ratio/region_mean": 0.1449021641165018, "reward_total_mean": 0.3530135750770569, "reward_meter_mean": 0.22241005301475525, "reward_meter_std": 0.26595112681388855, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8868178129196167, "reward_repeat_soft_std": 0.061087459325790405, "reward_judge_quality_mean": 0.5400000214576721, "reward_judge_quality_std": 0.14957083761692047, "reward_total_composite_mean": 0.3530135750770569, "reward_total_composite_std": 0.17210303246974945} {"timestamp_utc": "2026-04-13T09:00:32Z", "mode": "train", "global_step": 656, "epoch": 0.0658965344048217, "loss": 0.0221, "grad_norm": 12.217185020446777, "learning_rate": 8.015151515151515e-06, "num_tokens": 1160927.0, "completions/mean_length": 68.75, "completions/min_length": 53.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.75, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.5280009508132935, "rewards/meter/std": 0.4282487630844116, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8394860029220581, "rewards/repeat_soft/std": 0.11667388677597046, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.45864760875701904, "rewards/total_composite/std": 0.13028350472450256, "reward": 0.45864760875701904, "reward_std": 0.13028350472450256, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16322258114814758, "sampling/sampling_logp_difference/max": 1.7329931259155273, "sampling/importance_sampling_ratio/min": 0.17675456404685974, "sampling/importance_sampling_ratio/mean": 1.0246920585632324, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2120042070746422, "clip_ratio/low_mean": 0.05625782907009125, "clip_ratio/low_min": 0.05625782907009125, "clip_ratio/high_mean": 0.08202130533754826, "clip_ratio/high_max": 0.08202130533754826, "clip_ratio/region_mean": 0.1382791344076395, "reward_total_mean": 0.45864760875701904, "reward_meter_mean": 0.5280009508132935, "reward_meter_std": 0.4282487630844116, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8394860029220581, "reward_repeat_soft_std": 0.11667388677597046, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.45864760875701904, "reward_total_composite_std": 0.13028350472450256} {"timestamp_utc": "2026-04-13T09:00:38Z", "mode": "train", "global_step": 657, "epoch": 0.06599698643897539, "loss": -0.0175, "grad_norm": 14.234102249145508, "learning_rate": 8.012121212121214e-06, "num_tokens": 1162766.0, "completions/mean_length": 54.875, "completions/min_length": 43.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.875, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.5937342643737793, "rewards/meter/std": 0.2847641110420227, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9064621925354004, "rewards/repeat_soft/std": 0.09552107751369476, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.40473976731300354, "rewards/total_composite/std": 0.2639545202255249, "reward": 0.40473976731300354, "reward_std": 0.2639545202255249, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17224234342575073, "sampling/sampling_logp_difference/max": 2.0596227645874023, "sampling/importance_sampling_ratio/min": 0.12750205397605896, "sampling/importance_sampling_ratio/mean": 1.0342533588409424, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2091957330703735, "clip_ratio/low_mean": 0.05395756848156452, "clip_ratio/low_min": 0.05395756848156452, "clip_ratio/high_mean": 0.11370428279042244, "clip_ratio/high_max": 0.11370428279042244, "clip_ratio/region_mean": 0.16766185127198696, "reward_total_mean": 0.40473976731300354, "reward_meter_mean": 0.5937342643737793, "reward_meter_std": 0.2847641110420227, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9064621925354004, "reward_repeat_soft_std": 0.09552107751369476, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.40473976731300354, "reward_total_composite_std": 0.2639545202255249} {"timestamp_utc": "2026-04-13T09:00:45Z", "mode": "train", "global_step": 658, "epoch": 0.06609743847312909, "loss": 0.0362, "grad_norm": 15.805983543395996, "learning_rate": 8.00909090909091e-06, "num_tokens": 1164402.0, "completions/mean_length": 33.5, "completions/min_length": 31.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.5, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.32758235931396484, "rewards/meter/std": 0.3652130663394928, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9861834049224854, "rewards/repeat_soft/std": 0.030291561037302017, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.4663577079772949, "rewards/total_composite/std": 0.11732984334230423, "reward": 0.4663577079772949, "reward_std": 0.11732983589172363, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18540358543395996, "sampling/sampling_logp_difference/max": 1.5263197422027588, "sampling/importance_sampling_ratio/min": 0.21733404695987701, "sampling/importance_sampling_ratio/mean": 1.0284390449523926, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1702303737401962, "clip_ratio/low_mean": 0.11255082860589027, "clip_ratio/low_min": 0.11255082860589027, "clip_ratio/high_mean": 0.09381268778815866, "clip_ratio/high_max": 0.09381268778815866, "clip_ratio/region_mean": 0.20636351639404893, "reward_total_mean": 0.4663577079772949, "reward_meter_mean": 0.32758235931396484, "reward_meter_std": 0.3652130663394928, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9861834049224854, "reward_repeat_soft_std": 0.030291561037302017, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.4663577079772949, "reward_total_composite_std": 0.11732984334230423} {"timestamp_utc": "2026-04-13T09:00:50Z", "mode": "train", "global_step": 659, "epoch": 0.06619789050728277, "loss": 0.0285, "grad_norm": 13.327178955078125, "learning_rate": 8.006060606060607e-06, "num_tokens": 1165935.0, "completions/mean_length": 34.625, "completions/min_length": 31.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.625, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.7355386018753052, "rewards/meter/std": 0.33000099658966064, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7617665529251099, "rewards/repeat_soft/std": 0.1503032147884369, "rewards/judge_quality/mean": 0.38874998688697815, "rewards/judge_quality/std": 0.08675704896450043, "rewards/total_composite/mean": 0.4926672577857971, "rewards/total_composite/std": 0.09403003752231598, "reward": 0.4926672577857971, "reward_std": 0.09403003752231598, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16989152133464813, "sampling/sampling_logp_difference/max": 2.105639934539795, "sampling/importance_sampling_ratio/min": 0.12176772207021713, "sampling/importance_sampling_ratio/mean": 1.024625539779663, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1311260685324669, "clip_ratio/low_mean": 0.06997071579098701, "clip_ratio/low_min": 0.06997071579098701, "clip_ratio/high_mean": 0.06208881549537182, "clip_ratio/high_max": 0.06208881549537182, "clip_ratio/region_mean": 0.13205953128635883, "reward_total_mean": 0.4926672577857971, "reward_meter_mean": 0.7355386018753052, "reward_meter_std": 0.33000099658966064, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7617665529251099, "reward_repeat_soft_std": 0.1503032147884369, "reward_judge_quality_mean": 0.38874998688697815, "reward_judge_quality_std": 0.08675704896450043, "reward_total_composite_mean": 0.4926672577857971, "reward_total_composite_std": 0.09403003752231598} {"timestamp_utc": "2026-04-13T09:00:56Z", "mode": "train", "global_step": 660, "epoch": 0.06629834254143646, "loss": 0.0277, "grad_norm": 16.149572372436523, "learning_rate": 8.003030303030304e-06, "num_tokens": 1167513.0, "completions/mean_length": 30.25, "completions/min_length": 21.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.25, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.536858856678009, "rewards/meter/std": 0.34027794003486633, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9758846759796143, "rewards/repeat_soft/std": 0.035142626613378525, "rewards/judge_quality/mean": 0.7825000286102295, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.6137505769729614, "rewards/total_composite/std": 0.19050054252147675, "reward": 0.6137505769729614, "reward_std": 0.19050054252147675, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16642440855503082, "sampling/sampling_logp_difference/max": 1.8366745710372925, "sampling/importance_sampling_ratio/min": 0.15934644639492035, "sampling/importance_sampling_ratio/mean": 0.9946097731590271, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8458282127976418, "clip_ratio/low_mean": 0.05387118645012379, "clip_ratio/low_min": 0.05387118645012379, "clip_ratio/high_mean": 0.08504902245476842, "clip_ratio/high_max": 0.08504902245476842, "clip_ratio/region_mean": 0.1389202089048922, "reward_total_mean": 0.6137505769729614, "reward_meter_mean": 0.536858856678009, "reward_meter_std": 0.34027794003486633, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9758846759796143, "reward_repeat_soft_std": 0.035142626613378525, "reward_judge_quality_mean": 0.7825000286102295, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.6137505769729614, "reward_total_composite_std": 0.19050054252147675} {"timestamp_utc": "2026-04-13T09:01:02Z", "mode": "train", "global_step": 661, "epoch": 0.06639879457559016, "loss": 0.0367, "grad_norm": 21.243154525756836, "learning_rate": 8.000000000000001e-06, "num_tokens": 1168785.0, "completions/mean_length": 20.0, "completions/min_length": 17.0, "completions/max_length": 23.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 20.0, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 23.0, "rewards/meter/mean": 0.9752902388572693, "rewards/meter/std": 0.026702914386987686, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.617825984954834, "rewards/total_composite/std": 0.013858720660209656, "reward": 0.617825984954834, "reward_std": 0.01385872345417738, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2052830010652542, "sampling/sampling_logp_difference/max": 1.6482620239257812, "sampling/importance_sampling_ratio/min": 0.19238397479057312, "sampling/importance_sampling_ratio/mean": 1.0423879623413086, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2811303362250328, "clip_ratio/low_mean": 0.10084818955510855, "clip_ratio/low_min": 0.10084818955510855, "clip_ratio/high_mean": 0.04298563348129392, "clip_ratio/high_max": 0.04298563348129392, "clip_ratio/region_mean": 0.14383382303640246, "reward_total_mean": 0.617825984954834, "reward_meter_mean": 0.9752902388572693, "reward_meter_std": 0.026702914386987686, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.617825984954834, "reward_total_composite_std": 0.013858720660209656} {"timestamp_utc": "2026-04-13T09:01:08Z", "mode": "train", "global_step": 662, "epoch": 0.06649924660974385, "loss": 0.0195, "grad_norm": 15.285932540893555, "learning_rate": 7.996969696969697e-06, "num_tokens": 1170839.0, "completions/mean_length": 60.75, "completions/min_length": 46.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.75, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.8252530097961426, "rewards/meter/std": 0.16105490922927856, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.13363061845302582, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8301280736923218, "rewards/repeat_soft/std": 0.1212448701262474, "rewards/judge_quality/mean": 0.5324999690055847, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.5117673873901367, "rewards/total_composite/std": 0.2360391467809677, "reward": 0.5117673873901367, "reward_std": 0.2360391467809677, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1757470816373825, "sampling/sampling_logp_difference/max": 2.6512575149536133, "sampling/importance_sampling_ratio/min": 0.07056242227554321, "sampling/importance_sampling_ratio/mean": 1.018597960472107, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1257533058524132, "clip_ratio/low_mean": 0.05017806403338909, "clip_ratio/low_min": 0.05017806403338909, "clip_ratio/high_mean": 0.08945727348327637, "clip_ratio/high_max": 0.08945727348327637, "clip_ratio/region_mean": 0.13963533751666546, "reward_total_mean": 0.5117673873901367, "reward_meter_mean": 0.8252530097961426, "reward_meter_std": 0.16105490922927856, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.13363061845302582, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8301280736923218, "reward_repeat_soft_std": 0.1212448701262474, "reward_judge_quality_mean": 0.5324999690055847, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.5117673873901367, "reward_total_composite_std": 0.2360391467809677} {"timestamp_utc": "2026-04-13T09:01:20Z", "mode": "train", "global_step": 663, "epoch": 0.06659969864389755, "loss": -0.2039, "grad_norm": 2.632770299911499, "learning_rate": 7.993939393939396e-06, "num_tokens": 1173615.0, "completions/mean_length": 216.0, "completions/min_length": 112.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 117.33333587646484, "completions/min_terminated_length": 112.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.6868778467178345, "rewards/meter/std": 0.37099072337150574, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.1414213478565216, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9341126680374146, "rewards/repeat_soft/std": 0.03347941115498543, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.22025957703590393, "rewards/total_composite/mean": 0.42581695318222046, "rewards/total_composite/std": 0.28781354427337646, "reward": 0.42581695318222046, "reward_std": 0.28781354427337646, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13848777115345, "sampling/sampling_logp_difference/max": 3.1296353340148926, "sampling/importance_sampling_ratio/min": 0.04373374208807945, "sampling/importance_sampling_ratio/mean": 1.0213404893875122, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9202366322278976, "clip_ratio/low_mean": 0.01508620660752058, "clip_ratio/low_min": 0.01508620660752058, "clip_ratio/high_mean": 0.08940434642136097, "clip_ratio/high_max": 0.08940434642136097, "clip_ratio/region_mean": 0.10449055302888155, "reward_total_mean": 0.42581695318222046, "reward_meter_mean": 0.6868778467178345, "reward_meter_std": 0.37099072337150574, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.1414213478565216, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9341126680374146, "reward_repeat_soft_std": 0.03347941115498543, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.22025957703590393, "reward_total_composite_mean": 0.42581695318222046, "reward_total_composite_std": 0.28781354427337646} {"timestamp_utc": "2026-04-13T09:01:27Z", "mode": "train", "global_step": 664, "epoch": 0.06670015067805123, "loss": 0.0502, "grad_norm": 15.893335342407227, "learning_rate": 7.990909090909091e-06, "num_tokens": 1175225.0, "completions/mean_length": 35.25, "completions/min_length": 28.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.25, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.5234280824661255, "rewards/meter/std": 0.4268744885921478, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9757056832313538, "rewards/repeat_soft/std": 0.03202519565820694, "rewards/judge_quality/mean": 0.8575000166893005, "rewards/judge_quality/std": 0.176776722073555, "rewards/total_composite/mean": 0.653202474117279, "rewards/total_composite/std": 0.2629137635231018, "reward": 0.653202474117279, "reward_std": 0.2629137635231018, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1433289796113968, "sampling/sampling_logp_difference/max": 1.4580602645874023, "sampling/importance_sampling_ratio/min": 0.23268719017505646, "sampling/importance_sampling_ratio/mean": 1.003405213356018, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8421033546328545, "clip_ratio/low_mean": 0.06581361405551434, "clip_ratio/low_min": 0.06581361405551434, "clip_ratio/high_mean": 0.05746727483347058, "clip_ratio/high_max": 0.05746727483347058, "clip_ratio/region_mean": 0.12328088888898492, "reward_total_mean": 0.653202474117279, "reward_meter_mean": 0.5234280824661255, "reward_meter_std": 0.4268744885921478, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9757056832313538, "reward_repeat_soft_std": 0.03202519565820694, "reward_judge_quality_mean": 0.8575000166893005, "reward_judge_quality_std": 0.176776722073555, "reward_total_composite_mean": 0.653202474117279, "reward_total_composite_std": 0.2629137635231018} {"timestamp_utc": "2026-04-13T09:01:33Z", "mode": "train", "global_step": 665, "epoch": 0.06680060271220492, "loss": 0.0233, "grad_norm": 16.9578857421875, "learning_rate": 7.987878787878789e-06, "num_tokens": 1176813.0, "completions/mean_length": 39.5, "completions/min_length": 35.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.5, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.7170044183731079, "rewards/meter/std": 0.35669296979904175, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9703220725059509, "rewards/repeat_soft/std": 0.03601430356502533, "rewards/judge_quality/mean": 0.9200000166893005, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7743169665336609, "rewards/total_composite/std": 0.21356315910816193, "reward": 0.7743169665336609, "reward_std": 0.21356317400932312, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15803343057632446, "sampling/sampling_logp_difference/max": 1.2026662826538086, "sampling/importance_sampling_ratio/min": 0.300392210483551, "sampling/importance_sampling_ratio/mean": 1.0351260900497437, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0985603258013725, "clip_ratio/low_mean": 0.03826013393700123, "clip_ratio/low_min": 0.03826013393700123, "clip_ratio/high_mean": 0.08442175667732954, "clip_ratio/high_max": 0.08442175667732954, "clip_ratio/region_mean": 0.12268189061433077, "reward_total_mean": 0.7743169665336609, "reward_meter_mean": 0.7170044183731079, "reward_meter_std": 0.35669296979904175, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9703220725059509, "reward_repeat_soft_std": 0.03601430356502533, "reward_judge_quality_mean": 0.9200000166893005, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7743169665336609, "reward_total_composite_std": 0.21356315910816193} {"timestamp_utc": "2026-04-13T09:01:39Z", "mode": "train", "global_step": 666, "epoch": 0.06690105474635862, "loss": 0.0314, "grad_norm": 17.484010696411133, "learning_rate": 7.984848484848486e-06, "num_tokens": 1178291.0, "completions/mean_length": 29.75, "completions/min_length": 26.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.75, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.4946022927761078, "rewards/meter/std": 0.4055790603160858, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9355142116546631, "rewards/repeat_soft/std": 0.062372948974370956, "rewards/judge_quality/mean": 0.731249988079071, "rewards/judge_quality/std": 0.20131267607212067, "rewards/total_composite/mean": 0.5652487874031067, "rewards/total_composite/std": 0.19610926508903503, "reward": 0.5652487874031067, "reward_std": 0.19610926508903503, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19448567926883698, "sampling/sampling_logp_difference/max": 1.7358198165893555, "sampling/importance_sampling_ratio/min": 0.17625565826892853, "sampling/importance_sampling_ratio/mean": 0.9778200387954712, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1346284076571465, "clip_ratio/low_mean": 0.059890261851251125, "clip_ratio/low_min": 0.059890261851251125, "clip_ratio/high_mean": 0.0824202811345458, "clip_ratio/high_max": 0.0824202811345458, "clip_ratio/region_mean": 0.14231054298579693, "reward_total_mean": 0.5652487874031067, "reward_meter_mean": 0.4946022927761078, "reward_meter_std": 0.4055790603160858, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9355142116546631, "reward_repeat_soft_std": 0.062372948974370956, "reward_judge_quality_mean": 0.731249988079071, "reward_judge_quality_std": 0.20131267607212067, "reward_total_composite_mean": 0.5652487874031067, "reward_total_composite_std": 0.19610926508903503} {"timestamp_utc": "2026-04-13T09:01:45Z", "mode": "train", "global_step": 667, "epoch": 0.06700150678051231, "loss": 0.1176, "grad_norm": 31.319604873657227, "learning_rate": 7.981818181818183e-06, "num_tokens": 1179608.0, "completions/mean_length": 15.625, "completions/min_length": 13.0, "completions/max_length": 22.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 15.625, "completions/min_terminated_length": 13.0, "completions/max_terminated_length": 22.0, "rewards/meter/mean": 0.723710298538208, "rewards/meter/std": 0.35061559081077576, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.18616428971290588, "rewards/total_composite/mean": 0.5872097611427307, "rewards/total_composite/std": 0.17698632180690765, "reward": 0.5872097611427307, "reward_std": 0.17698630690574646, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17195676267147064, "sampling/sampling_logp_difference/max": 1.2872560024261475, "sampling/importance_sampling_ratio/min": 0.29236650466918945, "sampling/importance_sampling_ratio/mean": 1.0469970703125, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2363938093185425, "clip_ratio/low_mean": 0.16971050202846527, "clip_ratio/low_min": 0.16971050202846527, "clip_ratio/high_mean": 0.05444139428436756, "clip_ratio/high_max": 0.05444139428436756, "clip_ratio/region_mean": 0.22415189631283283, "reward_total_mean": 0.5872097611427307, "reward_meter_mean": 0.723710298538208, "reward_meter_std": 0.35061559081077576, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.18616428971290588, "reward_total_composite_mean": 0.5872097611427307, "reward_total_composite_std": 0.17698632180690765} {"timestamp_utc": "2026-04-13T09:01:50Z", "mode": "train", "global_step": 668, "epoch": 0.06710195881466599, "loss": 0.0449, "grad_norm": 17.066247940063477, "learning_rate": 7.978787878787879e-06, "num_tokens": 1181107.0, "completions/mean_length": 27.375, "completions/min_length": 24.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.375, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.64689040184021, "rewards/meter/std": 0.3717164993286133, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9223263263702393, "rewards/repeat_soft/std": 0.11539124697446823, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.10392305999994278, "rewards/total_composite/mean": 0.5421292781829834, "rewards/total_composite/std": 0.13657505810260773, "reward": 0.5421292781829834, "reward_std": 0.13657505810260773, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.163919135928154, "sampling/sampling_logp_difference/max": 1.8664093017578125, "sampling/importance_sampling_ratio/min": 0.15467806160449982, "sampling/importance_sampling_ratio/mean": 1.010160207748413, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8885713741183281, "clip_ratio/low_mean": 0.09868130926042795, "clip_ratio/low_min": 0.09868130926042795, "clip_ratio/high_mean": 0.07606082316488028, "clip_ratio/high_max": 0.07606082316488028, "clip_ratio/region_mean": 0.17474213242530823, "reward_total_mean": 0.5421292781829834, "reward_meter_mean": 0.64689040184021, "reward_meter_std": 0.3717164993286133, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9223263263702393, "reward_repeat_soft_std": 0.11539124697446823, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.10392305999994278, "reward_total_composite_mean": 0.5421292781829834, "reward_total_composite_std": 0.13657505810260773} {"timestamp_utc": "2026-04-13T09:01:56Z", "mode": "train", "global_step": 669, "epoch": 0.06720241084881969, "loss": 0.0342, "grad_norm": 25.62593650817871, "learning_rate": 7.975757575757576e-06, "num_tokens": 1182586.0, "completions/mean_length": 25.875, "completions/min_length": 23.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.875, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.46250832080841064, "rewards/meter/std": 0.3943176865577698, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9281853437423706, "rewards/repeat_soft/std": 0.05050138756632805, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.4678257703781128, "rewards/total_composite/std": 0.10863402485847473, "reward": 0.4678257703781128, "reward_std": 0.10863400250673294, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21301527321338654, "sampling/sampling_logp_difference/max": 2.5426676273345947, "sampling/importance_sampling_ratio/min": 0.07865629345178604, "sampling/importance_sampling_ratio/mean": 1.003975510597229, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2145325914025307, "clip_ratio/low_mean": 0.08599405828863382, "clip_ratio/low_min": 0.08599405828863382, "clip_ratio/high_mean": 0.07259339187294245, "clip_ratio/high_max": 0.07259339187294245, "clip_ratio/region_mean": 0.15858745016157627, "reward_total_mean": 0.4678257703781128, "reward_meter_mean": 0.46250832080841064, "reward_meter_std": 0.3943176865577698, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9281853437423706, "reward_repeat_soft_std": 0.05050138756632805, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.4678257703781128, "reward_total_composite_std": 0.10863402485847473} {"timestamp_utc": "2026-04-13T09:02:02Z", "mode": "train", "global_step": 670, "epoch": 0.06730286288297338, "loss": 0.032, "grad_norm": 8.41564655303955, "learning_rate": 7.972727272727273e-06, "num_tokens": 1184823.0, "completions/mean_length": 99.625, "completions/min_length": 93.0, "completions/max_length": 104.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 99.625, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.6294859647750854, "rewards/meter/std": 0.25513356924057007, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8679848909378052, "rewards/repeat_soft/std": 0.07367976009845734, "rewards/judge_quality/mean": 0.5325000286102295, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.5435631275177002, "rewards/total_composite/std": 0.09454849362373352, "reward": 0.5435631275177002, "reward_std": 0.09454847127199173, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16039173305034637, "sampling/sampling_logp_difference/max": 3.5806386470794678, "sampling/importance_sampling_ratio/min": 0.02785790152847767, "sampling/importance_sampling_ratio/mean": 1.0072435140609741, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1585305109620094, "clip_ratio/low_mean": 0.06731393747031689, "clip_ratio/low_min": 0.06731393747031689, "clip_ratio/high_mean": 0.06668214779347181, "clip_ratio/high_max": 0.06668214779347181, "clip_ratio/region_mean": 0.1339960852637887, "reward_total_mean": 0.5435631275177002, "reward_meter_mean": 0.6294859647750854, "reward_meter_std": 0.25513356924057007, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8679848909378052, "reward_repeat_soft_std": 0.07367976009845734, "reward_judge_quality_mean": 0.5325000286102295, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.5435631275177002, "reward_total_composite_std": 0.09454849362373352} {"timestamp_utc": "2026-04-13T09:02:08Z", "mode": "train", "global_step": 671, "epoch": 0.06740331491712707, "loss": 0.0513, "grad_norm": 21.132844924926758, "learning_rate": 7.96969696969697e-06, "num_tokens": 1186325.0, "completions/mean_length": 28.75, "completions/min_length": 26.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.75, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.7009587287902832, "rewards/meter/std": 0.40324610471725464, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9373697638511658, "rewards/repeat_soft/std": 0.0465090349316597, "rewards/judge_quality/mean": 0.8612500429153442, "rewards/judge_quality/std": 0.16617010533809662, "rewards/total_composite/mean": 0.7222044467926025, "rewards/total_composite/std": 0.2330915629863739, "reward": 0.7222044467926025, "reward_std": 0.2330915480852127, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1750107854604721, "sampling/sampling_logp_difference/max": 1.9450829029083252, "sampling/importance_sampling_ratio/min": 0.14297537505626678, "sampling/importance_sampling_ratio/mean": 1.0134085416793823, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.214572735130787, "clip_ratio/low_mean": 0.07826228905469179, "clip_ratio/low_min": 0.07826228905469179, "clip_ratio/high_mean": 0.07405641861259937, "clip_ratio/high_max": 0.07405641861259937, "clip_ratio/region_mean": 0.15231870766729116, "reward_total_mean": 0.7222044467926025, "reward_meter_mean": 0.7009587287902832, "reward_meter_std": 0.40324610471725464, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9373697638511658, "reward_repeat_soft_std": 0.0465090349316597, "reward_judge_quality_mean": 0.8612500429153442, "reward_judge_quality_std": 0.16617010533809662, "reward_total_composite_mean": 0.7222044467926025, "reward_total_composite_std": 0.2330915629863739} {"timestamp_utc": "2026-04-13T09:02:15Z", "mode": "train", "global_step": 672, "epoch": 0.06750376695128077, "loss": 0.0669, "grad_norm": 11.541581153869629, "learning_rate": 7.966666666666668e-06, "num_tokens": 1188303.0, "completions/mean_length": 79.25, "completions/min_length": 75.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.25, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.9345486164093018, "rewards/meter/std": 0.08608411997556686, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8871944546699524, "rewards/repeat_soft/std": 0.08728492259979248, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.6362572312355042, "rewards/total_composite/std": 0.10502653568983078, "reward": 0.6362572312355042, "reward_std": 0.10502654314041138, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17951758205890656, "sampling/sampling_logp_difference/max": 1.841355800628662, "sampling/importance_sampling_ratio/min": 0.17673450708389282, "sampling/importance_sampling_ratio/mean": 1.0232856273651123, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.519042745232582, "clip_ratio/low_mean": 0.09671382559463382, "clip_ratio/low_min": 0.09671382559463382, "clip_ratio/high_mean": 0.04755411110818386, "clip_ratio/high_max": 0.04755411110818386, "clip_ratio/region_mean": 0.14426793670281768, "reward_total_mean": 0.6362572312355042, "reward_meter_mean": 0.9345486164093018, "reward_meter_std": 0.08608411997556686, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8871944546699524, "reward_repeat_soft_std": 0.08728492259979248, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.6362572312355042, "reward_total_composite_std": 0.10502653568983078} {"timestamp_utc": "2026-04-13T09:02:26Z", "mode": "train", "global_step": 673, "epoch": 0.06760421898543445, "loss": -0.1507, "grad_norm": 3.268763542175293, "learning_rate": 7.963636363636365e-06, "num_tokens": 1190056.0, "completions/mean_length": 123.125, "completions/min_length": 60.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 67.5714340209961, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.8443180918693542, "rewards/meter/std": 0.2273014634847641, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8706649541854858, "rewards/repeat_soft/std": 0.17246706783771515, "rewards/judge_quality/mean": 0.38999998569488525, "rewards/judge_quality/std": 0.18845234811306, "rewards/total_composite/mean": 0.5125667452812195, "rewards/total_composite/std": 0.23342394828796387, "reward": 0.5125667452812195, "reward_std": 0.23342394828796387, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18086622655391693, "sampling/sampling_logp_difference/max": 1.1409001350402832, "sampling/importance_sampling_ratio/min": 0.31953126192092896, "sampling/importance_sampling_ratio/mean": 1.0215522050857544, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1888233423233032, "clip_ratio/low_mean": 0.022275641094893217, "clip_ratio/low_min": 0.022275641094893217, "clip_ratio/high_mean": 0.11251303926110268, "clip_ratio/high_max": 0.11251303926110268, "clip_ratio/region_mean": 0.1347886803559959, "reward_total_mean": 0.5125667452812195, "reward_meter_mean": 0.8443180918693542, "reward_meter_std": 0.2273014634847641, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8706649541854858, "reward_repeat_soft_std": 0.17246706783771515, "reward_judge_quality_mean": 0.38999998569488525, "reward_judge_quality_std": 0.18845234811306, "reward_total_composite_mean": 0.5125667452812195, "reward_total_composite_std": 0.23342394828796387} {"timestamp_utc": "2026-04-13T09:02:38Z", "mode": "train", "global_step": 674, "epoch": 0.06770467101958814, "loss": -0.1358, "grad_norm": 5.081193923950195, "learning_rate": 7.96060606060606e-06, "num_tokens": 1192216.0, "completions/mean_length": 147.0, "completions/min_length": 77.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 94.85714721679688, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 122.0, "rewards/meter/mean": 0.7837413549423218, "rewards/meter/std": 0.22885428369045258, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.2357022762298584, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.7908656597137451, "rewards/repeat_soft/std": 0.17838376760482788, "rewards/judge_quality/mean": 0.44875001907348633, "rewards/judge_quality/std": 0.2105392962694168, "rewards/total_composite/mean": 0.448514461517334, "rewards/total_composite/std": 0.2988634407520294, "reward": 0.448514461517334, "reward_std": 0.2988634407520294, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1417771726846695, "sampling/sampling_logp_difference/max": 1.746849536895752, "sampling/importance_sampling_ratio/min": 0.17432227730751038, "sampling/importance_sampling_ratio/mean": 1.0056918859481812, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7560857981443405, "clip_ratio/low_mean": 0.014044944196939468, "clip_ratio/low_min": 0.014044944196939468, "clip_ratio/high_mean": 0.08761232905089855, "clip_ratio/high_max": 0.08761232905089855, "clip_ratio/region_mean": 0.10165727324783802, "reward_total_mean": 0.448514461517334, "reward_meter_mean": 0.7837413549423218, "reward_meter_std": 0.22885428369045258, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.2357022762298584, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.7908656597137451, "reward_repeat_soft_std": 0.17838376760482788, "reward_judge_quality_mean": 0.44875001907348633, "reward_judge_quality_std": 0.2105392962694168, "reward_total_composite_mean": 0.448514461517334, "reward_total_composite_std": 0.2988634407520294} {"timestamp_utc": "2026-04-13T09:02:44Z", "mode": "train", "global_step": 675, "epoch": 0.06780512305374184, "loss": 0.0521, "grad_norm": 15.669044494628906, "learning_rate": 7.957575757575758e-06, "num_tokens": 1193848.0, "completions/mean_length": 35.0, "completions/min_length": 32.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.30102837085723877, "rewards/meter/std": 0.28132569789886475, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9912567138671875, "rewards/repeat_soft/std": 0.008987357839941978, "rewards/judge_quality/mean": 0.48625001311302185, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.44334959983825684, "rewards/total_composite/std": 0.08618506044149399, "reward": 0.44334959983825684, "reward_std": 0.08618505299091339, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15274398028850555, "sampling/sampling_logp_difference/max": 1.2224321365356445, "sampling/importance_sampling_ratio/min": 0.2945129871368408, "sampling/importance_sampling_ratio/mean": 1.026187777519226, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2333584874868393, "clip_ratio/low_mean": 0.06395565439015627, "clip_ratio/low_min": 0.06395565439015627, "clip_ratio/high_mean": 0.06867985613644123, "clip_ratio/high_max": 0.06867985613644123, "clip_ratio/region_mean": 0.1326355105265975, "reward_total_mean": 0.44334959983825684, "reward_meter_mean": 0.30102837085723877, "reward_meter_std": 0.28132569789886475, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9912567138671875, "reward_repeat_soft_std": 0.008987357839941978, "reward_judge_quality_mean": 0.48625001311302185, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.44334959983825684, "reward_total_composite_std": 0.08618506044149399} {"timestamp_utc": "2026-04-13T09:02:51Z", "mode": "train", "global_step": 676, "epoch": 0.06790557508789553, "loss": 0.1237, "grad_norm": 9.95625114440918, "learning_rate": 7.954545454545455e-06, "num_tokens": 1195821.0, "completions/mean_length": 74.625, "completions/min_length": 56.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.625, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.7109752893447876, "rewards/meter/std": 0.34340932965278625, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6913044452667236, "rewards/repeat_soft/std": 0.18606877326965332, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.5181210041046143, "rewards/total_composite/std": 0.12576277554035187, "reward": 0.5181210041046143, "reward_std": 0.12576277554035187, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12566405534744263, "sampling/sampling_logp_difference/max": 1.4823908805847168, "sampling/importance_sampling_ratio/min": 0.22709408402442932, "sampling/importance_sampling_ratio/mean": 1.0080106258392334, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6787445619702339, "clip_ratio/low_mean": 0.05535984970629215, "clip_ratio/low_min": 0.05535984970629215, "clip_ratio/high_mean": 0.06927759945392609, "clip_ratio/high_max": 0.06927759945392609, "clip_ratio/region_mean": 0.12463744916021824, "reward_total_mean": 0.5181210041046143, "reward_meter_mean": 0.7109752893447876, "reward_meter_std": 0.34340932965278625, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6913044452667236, "reward_repeat_soft_std": 0.18606877326965332, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.5181210041046143, "reward_total_composite_std": 0.12576277554035187} {"timestamp_utc": "2026-04-13T09:02:57Z", "mode": "train", "global_step": 677, "epoch": 0.06800602712204921, "loss": 0.0415, "grad_norm": 15.976642608642578, "learning_rate": 7.951515151515152e-06, "num_tokens": 1197544.0, "completions/mean_length": 38.375, "completions/min_length": 32.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.375, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.3758412003517151, "rewards/meter/std": 0.3048185706138611, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9766628742218018, "rewards/repeat_soft/std": 0.03151514008641243, "rewards/judge_quality/mean": 0.5824999809265137, "rewards/judge_quality/std": 0.23260943591594696, "rewards/total_composite/mean": 0.4973105490207672, "rewards/total_composite/std": 0.15020503103733063, "reward": 0.4973105490207672, "reward_std": 0.15020503103733063, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16625650227069855, "sampling/sampling_logp_difference/max": 1.1946462392807007, "sampling/importance_sampling_ratio/min": 0.30281105637550354, "sampling/importance_sampling_ratio/mean": 1.0364453792572021, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3895529732108116, "clip_ratio/low_mean": 0.09560188185423613, "clip_ratio/low_min": 0.09560188185423613, "clip_ratio/high_mean": 0.02656250074505806, "clip_ratio/high_max": 0.02656250074505806, "clip_ratio/region_mean": 0.12216438259929419, "reward_total_mean": 0.4973105490207672, "reward_meter_mean": 0.3758412003517151, "reward_meter_std": 0.3048185706138611, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9766628742218018, "reward_repeat_soft_std": 0.03151514008641243, "reward_judge_quality_mean": 0.5824999809265137, "reward_judge_quality_std": 0.23260943591594696, "reward_total_composite_mean": 0.4973105490207672, "reward_total_composite_std": 0.15020503103733063} {"timestamp_utc": "2026-04-13T09:03:03Z", "mode": "train", "global_step": 678, "epoch": 0.06810647915620291, "loss": 0.1921, "grad_norm": 34.70186996459961, "learning_rate": 7.948484848484848e-06, "num_tokens": 1199079.0, "completions/mean_length": 37.875, "completions/min_length": 26.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.875, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.29696419835090637, "rewards/meter/std": 0.3232930302619934, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.972385048866272, "rewards/repeat_soft/std": 0.025973200798034668, "rewards/judge_quality/mean": 0.7400000095367432, "rewards/judge_quality/std": 0.24859607219696045, "rewards/total_composite/mean": 0.46499231457710266, "rewards/total_composite/std": 0.10828929394483566, "reward": 0.46499231457710266, "reward_std": 0.10828928649425507, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19013623893260956, "sampling/sampling_logp_difference/max": 2.3498706817626953, "sampling/importance_sampling_ratio/min": 0.0953814908862114, "sampling/importance_sampling_ratio/mean": 0.9908564686775208, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4125063568353653, "clip_ratio/low_mean": 0.0727268448099494, "clip_ratio/low_min": 0.0727268448099494, "clip_ratio/high_mean": 0.06334841810166836, "clip_ratio/high_max": 0.06334841810166836, "clip_ratio/region_mean": 0.13607526291161776, "reward_total_mean": 0.46499231457710266, "reward_meter_mean": 0.29696419835090637, "reward_meter_std": 0.3232930302619934, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.972385048866272, "reward_repeat_soft_std": 0.025973200798034668, "reward_judge_quality_mean": 0.7400000095367432, "reward_judge_quality_std": 0.24859607219696045, "reward_total_composite_mean": 0.46499231457710266, "reward_total_composite_std": 0.10828929394483566} {"timestamp_utc": "2026-04-13T09:03:09Z", "mode": "train", "global_step": 679, "epoch": 0.0682069311903566, "loss": 0.1029, "grad_norm": 13.571893692016602, "learning_rate": 7.945454545454547e-06, "num_tokens": 1200869.0, "completions/mean_length": 40.75, "completions/min_length": 29.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.75, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.5557988882064819, "rewards/meter/std": 0.353110671043396, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.900698184967041, "rewards/repeat_soft/std": 0.07676287740468979, "rewards/judge_quality/mean": 0.5600000619888306, "rewards/judge_quality/std": 0.31550416350364685, "rewards/total_composite/mean": 0.5425010919570923, "rewards/total_composite/std": 0.178108811378479, "reward": 0.5425010919570923, "reward_std": 0.178108811378479, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1761329472064972, "sampling/sampling_logp_difference/max": 1.5521812438964844, "sampling/importance_sampling_ratio/min": 0.21178551018238068, "sampling/importance_sampling_ratio/mean": 1.023783802986145, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.722318395972252, "clip_ratio/low_mean": 0.08291913568973541, "clip_ratio/low_min": 0.08291913568973541, "clip_ratio/high_mean": 0.04797446262091398, "clip_ratio/high_max": 0.04797446262091398, "clip_ratio/region_mean": 0.1308935983106494, "reward_total_mean": 0.5425010919570923, "reward_meter_mean": 0.5557988882064819, "reward_meter_std": 0.353110671043396, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.900698184967041, "reward_repeat_soft_std": 0.07676287740468979, "reward_judge_quality_mean": 0.5600000619888306, "reward_judge_quality_std": 0.31550416350364685, "reward_total_composite_mean": 0.5425010919570923, "reward_total_composite_std": 0.178108811378479} {"timestamp_utc": "2026-04-13T09:03:15Z", "mode": "train", "global_step": 680, "epoch": 0.0683073832245103, "loss": 0.0217, "grad_norm": 19.444847106933594, "learning_rate": 7.942424242424242e-06, "num_tokens": 1202320.0, "completions/mean_length": 33.375, "completions/min_length": 32.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.375, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9708325266838074, "rewards/meter/std": 0.024752020835876465, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9009986519813538, "rewards/repeat_soft/std": 0.07457141578197479, "rewards/judge_quality/mean": 0.59375, "rewards/judge_quality/std": 0.2775370180606842, "rewards/total_composite/mean": 0.7108228206634521, "rewards/total_composite/std": 0.18475474417209625, "reward": 0.7108228206634521, "reward_std": 0.18475474417209625, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17875127494335175, "sampling/sampling_logp_difference/max": 3.197706699371338, "sampling/importance_sampling_ratio/min": 0.04085579141974449, "sampling/importance_sampling_ratio/mean": 1.0433377027511597, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3223681598901749, "clip_ratio/low_mean": 0.11509859748184681, "clip_ratio/low_min": 0.11509859748184681, "clip_ratio/high_mean": 0.043231177143752575, "clip_ratio/high_max": 0.043231177143752575, "clip_ratio/region_mean": 0.15832977462559938, "reward_total_mean": 0.7108228206634521, "reward_meter_mean": 0.9708325266838074, "reward_meter_std": 0.024752020835876465, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9009986519813538, "reward_repeat_soft_std": 0.07457141578197479, "reward_judge_quality_mean": 0.59375, "reward_judge_quality_std": 0.2775370180606842, "reward_total_composite_mean": 0.7108228206634521, "reward_total_composite_std": 0.18475474417209625} {"timestamp_utc": "2026-04-13T09:03:20Z", "mode": "train", "global_step": 681, "epoch": 0.068407835258664, "loss": -0.0315, "grad_norm": 13.85059928894043, "learning_rate": 7.93939393939394e-06, "num_tokens": 1204056.0, "completions/mean_length": 40.0, "completions/min_length": 34.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.0, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.8067896962165833, "rewards/meter/std": 0.33366015553474426, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9807987213134766, "rewards/repeat_soft/std": 0.023438427597284317, "rewards/judge_quality/mean": 0.6112500429153442, "rewards/judge_quality/std": 0.25587037205696106, "rewards/total_composite/mean": 0.6513296365737915, "rewards/total_composite/std": 0.18861781060695648, "reward": 0.6513296365737915, "reward_std": 0.1886177957057953, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17290246486663818, "sampling/sampling_logp_difference/max": 1.6026787757873535, "sampling/importance_sampling_ratio/min": 0.20135639607906342, "sampling/importance_sampling_ratio/mean": 1.0108685493469238, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.389533892273903, "clip_ratio/low_mean": 0.10763272503390908, "clip_ratio/low_min": 0.10763272503390908, "clip_ratio/high_mean": 0.02724359044805169, "clip_ratio/high_max": 0.02724359044805169, "clip_ratio/region_mean": 0.13487631548196077, "reward_total_mean": 0.6513296365737915, "reward_meter_mean": 0.8067896962165833, "reward_meter_std": 0.33366015553474426, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9807987213134766, "reward_repeat_soft_std": 0.023438427597284317, "reward_judge_quality_mean": 0.6112500429153442, "reward_judge_quality_std": 0.25587037205696106, "reward_total_composite_mean": 0.6513296365737915, "reward_total_composite_std": 0.18861781060695648} {"timestamp_utc": "2026-04-13T09:03:26Z", "mode": "train", "global_step": 682, "epoch": 0.06850828729281767, "loss": 0.0067, "grad_norm": 15.246498107910156, "learning_rate": 7.936363636363637e-06, "num_tokens": 1205804.0, "completions/mean_length": 43.5, "completions/min_length": 39.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.5, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.7407575845718384, "rewards/meter/std": 0.2606121599674225, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9272711873054504, "rewards/repeat_soft/std": 0.04879491776227951, "rewards/judge_quality/mean": 0.5862500071525574, "rewards/judge_quality/std": 0.2353682667016983, "rewards/total_composite/mean": 0.6343569755554199, "rewards/total_composite/std": 0.175083190202713, "reward": 0.6343569755554199, "reward_std": 0.1750832051038742, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15224520862102509, "sampling/sampling_logp_difference/max": 1.8238554000854492, "sampling/importance_sampling_ratio/min": 0.16140228509902954, "sampling/importance_sampling_ratio/mean": 1.0225683450698853, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8424210101366043, "clip_ratio/low_mean": 0.08994126692414284, "clip_ratio/low_min": 0.08994126692414284, "clip_ratio/high_mean": 0.04844512231647968, "clip_ratio/high_max": 0.04844512231647968, "clip_ratio/region_mean": 0.13838638924062252, "reward_total_mean": 0.6343569755554199, "reward_meter_mean": 0.7407575845718384, "reward_meter_std": 0.2606121599674225, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9272711873054504, "reward_repeat_soft_std": 0.04879491776227951, "reward_judge_quality_mean": 0.5862500071525574, "reward_judge_quality_std": 0.2353682667016983, "reward_total_composite_mean": 0.6343569755554199, "reward_total_composite_std": 0.175083190202713} {"timestamp_utc": "2026-04-13T09:03:32Z", "mode": "train", "global_step": 683, "epoch": 0.06860873932697137, "loss": -0.0441, "grad_norm": 19.87747573852539, "learning_rate": 7.933333333333334e-06, "num_tokens": 1207392.0, "completions/mean_length": 38.5, "completions/min_length": 32.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.5, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.7616804838180542, "rewards/meter/std": 0.3156481087207794, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8836441040039062, "rewards/repeat_soft/std": 0.06109395623207092, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.5482738018035889, "rewards/total_composite/std": 0.09274234622716904, "reward": 0.5482738018035889, "reward_std": 0.09274233877658844, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1745234578847885, "sampling/sampling_logp_difference/max": 2.012331008911133, "sampling/importance_sampling_ratio/min": 0.1336767077445984, "sampling/importance_sampling_ratio/mean": 1.0172439813613892, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2272211387753487, "clip_ratio/low_mean": 0.030560662038624287, "clip_ratio/low_min": 0.030560662038624287, "clip_ratio/high_mean": 0.14728633873164654, "clip_ratio/high_max": 0.14728633873164654, "clip_ratio/region_mean": 0.17784700077027082, "reward_total_mean": 0.5482738018035889, "reward_meter_mean": 0.7616804838180542, "reward_meter_std": 0.3156481087207794, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8836441040039062, "reward_repeat_soft_std": 0.06109395623207092, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.5482738018035889, "reward_total_composite_std": 0.09274234622716904} {"timestamp_utc": "2026-04-13T09:03:38Z", "mode": "train", "global_step": 684, "epoch": 0.06870919136112506, "loss": 0.0117, "grad_norm": 15.160962104797363, "learning_rate": 7.930303030303031e-06, "num_tokens": 1208791.0, "completions/mean_length": 30.875, "completions/min_length": 27.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.875, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9185324907302856, "rewards/meter/std": 0.1610766500234604, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8485807776451111, "rewards/repeat_soft/std": 0.08786656707525253, "rewards/judge_quality/mean": 0.5862500071525574, "rewards/judge_quality/std": 0.2822834253311157, "rewards/total_composite/mean": 0.6363656520843506, "rewards/total_composite/std": 0.3035244345664978, "reward": 0.6363656520843506, "reward_std": 0.3035244345664978, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16926787793636322, "sampling/sampling_logp_difference/max": 1.9378900527954102, "sampling/importance_sampling_ratio/min": 0.14400747418403625, "sampling/importance_sampling_ratio/mean": 1.0151031017303467, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9692705720663071, "clip_ratio/low_mean": 0.06495348457247019, "clip_ratio/low_min": 0.06495348457247019, "clip_ratio/high_mean": 0.037830933928489685, "clip_ratio/high_max": 0.037830933928489685, "clip_ratio/region_mean": 0.10278441850095987, "reward_total_mean": 0.6363656520843506, "reward_meter_mean": 0.9185324907302856, "reward_meter_std": 0.1610766500234604, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8485807776451111, "reward_repeat_soft_std": 0.08786656707525253, "reward_judge_quality_mean": 0.5862500071525574, "reward_judge_quality_std": 0.2822834253311157, "reward_total_composite_mean": 0.6363656520843506, "reward_total_composite_std": 0.3035244345664978} {"timestamp_utc": "2026-04-13T09:03:44Z", "mode": "train", "global_step": 685, "epoch": 0.06880964339527876, "loss": 0.0278, "grad_norm": 15.482486724853516, "learning_rate": 7.927272727272729e-06, "num_tokens": 1210396.0, "completions/mean_length": 35.625, "completions/min_length": 33.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.625, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.5370067358016968, "rewards/meter/std": 0.42286112904548645, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7139323949813843, "rewards/repeat_soft/std": 0.21135301887989044, "rewards/judge_quality/mean": 0.4137499928474426, "rewards/judge_quality/std": 0.06781013309955597, "rewards/total_composite/mean": 0.4486607313156128, "rewards/total_composite/std": 0.10594851523637772, "reward": 0.4486607313156128, "reward_std": 0.10594853013753891, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16646844148635864, "sampling/sampling_logp_difference/max": 1.4042901992797852, "sampling/importance_sampling_ratio/min": 0.2455412894487381, "sampling/importance_sampling_ratio/mean": 1.025395154953003, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1219540163874626, "clip_ratio/low_mean": 0.06411921512335539, "clip_ratio/low_min": 0.06411921512335539, "clip_ratio/high_mean": 0.0721204336732626, "clip_ratio/high_max": 0.0721204336732626, "clip_ratio/region_mean": 0.13623964879661798, "reward_total_mean": 0.4486607313156128, "reward_meter_mean": 0.5370067358016968, "reward_meter_std": 0.42286112904548645, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7139323949813843, "reward_repeat_soft_std": 0.21135301887989044, "reward_judge_quality_mean": 0.4137499928474426, "reward_judge_quality_std": 0.06781013309955597, "reward_total_composite_mean": 0.4486607313156128, "reward_total_composite_std": 0.10594851523637772} {"timestamp_utc": "2026-04-13T09:03:50Z", "mode": "train", "global_step": 686, "epoch": 0.06891009542943245, "loss": -0.0335, "grad_norm": 15.697717666625977, "learning_rate": 7.924242424242426e-06, "num_tokens": 1211919.0, "completions/mean_length": 32.375, "completions/min_length": 26.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.375, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.3621957302093506, "rewards/meter/std": 0.3310295343399048, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.940528154373169, "rewards/repeat_soft/std": 0.11192068457603455, "rewards/judge_quality/mean": 0.5199999809265137, "rewards/judge_quality/std": 0.19272483885288239, "rewards/total_composite/mean": 0.4836890399456024, "rewards/total_composite/std": 0.1959802210330963, "reward": 0.4836890399456024, "reward_std": 0.1959802210330963, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17518004775047302, "sampling/sampling_logp_difference/max": 1.3122613430023193, "sampling/importance_sampling_ratio/min": 0.2692105770111084, "sampling/importance_sampling_ratio/mean": 1.0158535242080688, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4639041870832443, "clip_ratio/low_mean": 0.08152802241966128, "clip_ratio/low_min": 0.08152802241966128, "clip_ratio/high_mean": 0.07402259670197964, "clip_ratio/high_max": 0.07402259670197964, "clip_ratio/region_mean": 0.15555061912164092, "reward_total_mean": 0.4836890399456024, "reward_meter_mean": 0.3621957302093506, "reward_meter_std": 0.3310295343399048, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.940528154373169, "reward_repeat_soft_std": 0.11192068457603455, "reward_judge_quality_mean": 0.5199999809265137, "reward_judge_quality_std": 0.19272483885288239, "reward_total_composite_mean": 0.4836890399456024, "reward_total_composite_std": 0.1959802210330963} {"timestamp_utc": "2026-04-13T09:03:56Z", "mode": "train", "global_step": 687, "epoch": 0.06901054746358613, "loss": -0.0076, "grad_norm": 16.958520889282227, "learning_rate": 7.921212121212122e-06, "num_tokens": 1213395.0, "completions/mean_length": 31.5, "completions/min_length": 27.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.5, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.7834794521331787, "rewards/meter/std": 0.3169131278991699, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9552586078643799, "rewards/repeat_soft/std": 0.044948361814022064, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.6794179677963257, "rewards/total_composite/std": 0.21409960091114044, "reward": 0.6794179677963257, "reward_std": 0.21409958600997925, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16569404304027557, "sampling/sampling_logp_difference/max": 1.0826793909072876, "sampling/importance_sampling_ratio/min": 0.33868685364723206, "sampling/importance_sampling_ratio/mean": 1.0096803903579712, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0858681723475456, "clip_ratio/low_mean": 0.10208950191736221, "clip_ratio/low_min": 0.10208950191736221, "clip_ratio/high_mean": 0.0692730974406004, "clip_ratio/high_max": 0.0692730974406004, "clip_ratio/region_mean": 0.1713625993579626, "reward_total_mean": 0.6794179677963257, "reward_meter_mean": 0.7834794521331787, "reward_meter_std": 0.3169131278991699, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9552586078643799, "reward_repeat_soft_std": 0.044948361814022064, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.6794179677963257, "reward_total_composite_std": 0.21409960091114044} {"timestamp_utc": "2026-04-13T09:04:08Z", "mode": "train", "global_step": 688, "epoch": 0.06911099949773983, "loss": -0.053, "grad_norm": 4.748079776763916, "learning_rate": 7.918181818181819e-06, "num_tokens": 1214873.0, "completions/mean_length": 94.75, "completions/min_length": 31.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 35.142860412597656, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.8547170162200928, "rewards/meter/std": 0.2940731942653656, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8981757164001465, "rewards/repeat_soft/std": 0.10416964441537857, "rewards/judge_quality/mean": 0.5724999904632568, "rewards/judge_quality/std": 0.31702861189842224, "rewards/total_composite/mean": 0.672359824180603, "rewards/total_composite/std": 0.2378748655319214, "reward": 0.672359824180603, "reward_std": 0.2378748655319214, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17198236286640167, "sampling/sampling_logp_difference/max": 1.9869155883789062, "sampling/importance_sampling_ratio/min": 0.13711769878864288, "sampling/importance_sampling_ratio/mean": 1.0300874710083008, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2422862350940704, "clip_ratio/low_mean": 0.056661609560251236, "clip_ratio/low_min": 0.056661609560251236, "clip_ratio/high_mean": 0.03858352266252041, "clip_ratio/high_max": 0.03858352266252041, "clip_ratio/region_mean": 0.09524513222277164, "reward_total_mean": 0.672359824180603, "reward_meter_mean": 0.8547170162200928, "reward_meter_std": 0.2940731942653656, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8981757164001465, "reward_repeat_soft_std": 0.10416964441537857, "reward_judge_quality_mean": 0.5724999904632568, "reward_judge_quality_std": 0.31702861189842224, "reward_total_composite_mean": 0.672359824180603, "reward_total_composite_std": 0.2378748655319214} {"timestamp_utc": "2026-04-13T09:04:16Z", "mode": "train", "global_step": 689, "epoch": 0.06921145153189352, "loss": 0.0227, "grad_norm": 14.012819290161133, "learning_rate": 7.915151515151516e-06, "num_tokens": 1216639.0, "completions/mean_length": 53.75, "completions/min_length": 50.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.75, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.4307352900505066, "rewards/meter/std": 0.34267833828926086, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9898139238357544, "rewards/repeat_soft/std": 0.011434405110776424, "rewards/judge_quality/mean": 0.5774999856948853, "rewards/judge_quality/std": 0.1527603417634964, "rewards/total_composite/mean": 0.48308122158050537, "rewards/total_composite/std": 0.2503104507923126, "reward": 0.48308122158050537, "reward_std": 0.25031042098999023, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2066158801317215, "sampling/sampling_logp_difference/max": 1.6739482879638672, "sampling/importance_sampling_ratio/min": 0.18750527501106262, "sampling/importance_sampling_ratio/mean": 1.0345228910446167, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4963724464178085, "clip_ratio/low_mean": 0.11144568957388401, "clip_ratio/low_min": 0.11144568957388401, "clip_ratio/high_mean": 0.05153766833245754, "clip_ratio/high_max": 0.05153766833245754, "clip_ratio/region_mean": 0.16298335790634155, "reward_total_mean": 0.48308122158050537, "reward_meter_mean": 0.4307352900505066, "reward_meter_std": 0.34267833828926086, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9898139238357544, "reward_repeat_soft_std": 0.011434405110776424, "reward_judge_quality_mean": 0.5774999856948853, "reward_judge_quality_std": 0.1527603417634964, "reward_total_composite_mean": 0.48308122158050537, "reward_total_composite_std": 0.2503104507923126} {"timestamp_utc": "2026-04-13T09:04:23Z", "mode": "train", "global_step": 690, "epoch": 0.06931190356604722, "loss": 0.0229, "grad_norm": 26.791959762573242, "learning_rate": 7.912121212121213e-06, "num_tokens": 1217928.0, "completions/mean_length": 17.125, "completions/min_length": 15.0, "completions/max_length": 22.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 17.125, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 22.0, "rewards/meter/mean": 0.24231413006782532, "rewards/meter/std": 0.3449556231498718, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9539605379104614, "rewards/repeat_soft/std": 0.022088520228862762, "rewards/judge_quality/mean": 0.7362500429153442, "rewards/judge_quality/std": 0.25376805663108826, "rewards/total_composite/mean": 0.47506770491600037, "rewards/total_composite/std": 0.20589597523212433, "reward": 0.47506770491600037, "reward_std": 0.20589596033096313, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15316243469715118, "sampling/sampling_logp_difference/max": 1.390833854675293, "sampling/importance_sampling_ratio/min": 0.2488677054643631, "sampling/importance_sampling_ratio/mean": 1.047309398651123, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.260609120130539, "clip_ratio/low_mean": 0.1275846753269434, "clip_ratio/low_min": 0.1275846753269434, "clip_ratio/high_mean": 0.028594771400094032, "clip_ratio/high_max": 0.028594771400094032, "clip_ratio/region_mean": 0.15617944672703743, "reward_total_mean": 0.47506770491600037, "reward_meter_mean": 0.24231413006782532, "reward_meter_std": 0.3449556231498718, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9539605379104614, "reward_repeat_soft_std": 0.022088520228862762, "reward_judge_quality_mean": 0.7362500429153442, "reward_judge_quality_std": 0.25376805663108826, "reward_total_composite_mean": 0.47506770491600037, "reward_total_composite_std": 0.20589597523212433} {"timestamp_utc": "2026-04-13T09:04:29Z", "mode": "train", "global_step": 691, "epoch": 0.0694123556002009, "loss": -0.0811, "grad_norm": 16.744775772094727, "learning_rate": 7.909090909090909e-06, "num_tokens": 1219267.0, "completions/mean_length": 31.375, "completions/min_length": 26.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.375, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.8912780284881592, "rewards/meter/std": 0.22956277430057526, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7346838712692261, "rewards/repeat_soft/std": 0.17986097931861877, "rewards/judge_quality/mean": 0.6225000023841858, "rewards/judge_quality/std": 0.24656209349632263, "rewards/total_composite/mean": 0.6554957628250122, "rewards/total_composite/std": 0.16776107251644135, "reward": 0.6554957628250122, "reward_std": 0.16776105761528015, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1292048692703247, "sampling/sampling_logp_difference/max": 1.3336560726165771, "sampling/importance_sampling_ratio/min": 0.2635120749473572, "sampling/importance_sampling_ratio/mean": 1.0278366804122925, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9303603619337082, "clip_ratio/low_mean": 0.08603221224620938, "clip_ratio/low_min": 0.08603221224620938, "clip_ratio/high_mean": 0.0324730109423399, "clip_ratio/high_max": 0.0324730109423399, "clip_ratio/region_mean": 0.11850522318854928, "reward_total_mean": 0.6554957628250122, "reward_meter_mean": 0.8912780284881592, "reward_meter_std": 0.22956277430057526, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7346838712692261, "reward_repeat_soft_std": 0.17986097931861877, "reward_judge_quality_mean": 0.6225000023841858, "reward_judge_quality_std": 0.24656209349632263, "reward_total_composite_mean": 0.6554957628250122, "reward_total_composite_std": 0.16776107251644135} {"timestamp_utc": "2026-04-13T09:04:36Z", "mode": "train", "global_step": 692, "epoch": 0.06951280763435459, "loss": -0.0282, "grad_norm": 14.995078086853027, "learning_rate": 7.906060606060608e-06, "num_tokens": 1220877.0, "completions/mean_length": 39.25, "completions/min_length": 33.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.25, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.6253436803817749, "rewards/meter/std": 0.3929384648799896, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8773505687713623, "rewards/repeat_soft/std": 0.11419042199850082, "rewards/judge_quality/mean": 0.4087499976158142, "rewards/judge_quality/std": 0.15412774682044983, "rewards/total_composite/mean": 0.4678286910057068, "rewards/total_composite/std": 0.24292142689228058, "reward": 0.4678286910057068, "reward_std": 0.24292141199111938, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1778692752122879, "sampling/sampling_logp_difference/max": 1.5066614151000977, "sampling/importance_sampling_ratio/min": 0.22164873778820038, "sampling/importance_sampling_ratio/mean": 1.030326008796692, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4388520568609238, "clip_ratio/low_mean": 0.05966448783874512, "clip_ratio/low_min": 0.05966448783874512, "clip_ratio/high_mean": 0.11110597476363182, "clip_ratio/high_max": 0.11110597476363182, "clip_ratio/region_mean": 0.17077046260237694, "reward_total_mean": 0.4678286910057068, "reward_meter_mean": 0.6253436803817749, "reward_meter_std": 0.3929384648799896, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8773505687713623, "reward_repeat_soft_std": 0.11419042199850082, "reward_judge_quality_mean": 0.4087499976158142, "reward_judge_quality_std": 0.15412774682044983, "reward_total_composite_mean": 0.4678286910057068, "reward_total_composite_std": 0.24292142689228058} {"timestamp_utc": "2026-04-13T09:04:43Z", "mode": "train", "global_step": 693, "epoch": 0.06961325966850829, "loss": 0.006, "grad_norm": 11.027209281921387, "learning_rate": 7.903030303030303e-06, "num_tokens": 1222687.0, "completions/mean_length": 61.25, "completions/min_length": 52.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.25, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.76621413230896, "rewards/meter/std": 0.3152752220630646, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8095709085464478, "rewards/repeat_soft/std": 0.14905677735805511, "rewards/judge_quality/mean": 0.38624998927116394, "rewards/judge_quality/std": 0.09545940905809402, "rewards/total_composite/mean": 0.5088721513748169, "rewards/total_composite/std": 0.10416708141565323, "reward": 0.5088721513748169, "reward_std": 0.10416708141565323, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1528916209936142, "sampling/sampling_logp_difference/max": 1.7419815063476562, "sampling/importance_sampling_ratio/min": 0.17517295479774475, "sampling/importance_sampling_ratio/mean": 1.021276831626892, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1214285790920258, "clip_ratio/low_mean": 0.041261966340243816, "clip_ratio/low_min": 0.041261966340243816, "clip_ratio/high_mean": 0.10720415972173214, "clip_ratio/high_max": 0.10720415972173214, "clip_ratio/region_mean": 0.14846612606197596, "reward_total_mean": 0.5088721513748169, "reward_meter_mean": 0.76621413230896, "reward_meter_std": 0.3152752220630646, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8095709085464478, "reward_repeat_soft_std": 0.14905677735805511, "reward_judge_quality_mean": 0.38624998927116394, "reward_judge_quality_std": 0.09545940905809402, "reward_total_composite_mean": 0.5088721513748169, "reward_total_composite_std": 0.10416708141565323} {"timestamp_utc": "2026-04-13T09:04:50Z", "mode": "train", "global_step": 694, "epoch": 0.06971371170266198, "loss": 0.0897, "grad_norm": 14.673514366149902, "learning_rate": 7.9e-06, "num_tokens": 1224270.0, "completions/mean_length": 38.875, "completions/min_length": 29.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.875, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.5953438878059387, "rewards/meter/std": 0.3709369897842407, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9078002572059631, "rewards/repeat_soft/std": 0.08058351278305054, "rewards/judge_quality/mean": 0.5562499761581421, "rewards/judge_quality/std": 0.2249404937028885, "rewards/total_composite/mean": 0.5727308988571167, "rewards/total_composite/std": 0.20136988162994385, "reward": 0.5727308988571167, "reward_std": 0.20136989653110504, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11624979227781296, "sampling/sampling_logp_difference/max": 1.484346866607666, "sampling/importance_sampling_ratio/min": 0.22665032744407654, "sampling/importance_sampling_ratio/mean": 1.0144704580307007, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9147335216403008, "clip_ratio/low_mean": 0.04443535767495632, "clip_ratio/low_min": 0.04443535767495632, "clip_ratio/high_mean": 0.08160406164824963, "clip_ratio/high_max": 0.08160406164824963, "clip_ratio/region_mean": 0.12603941932320595, "reward_total_mean": 0.5727308988571167, "reward_meter_mean": 0.5953438878059387, "reward_meter_std": 0.3709369897842407, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9078002572059631, "reward_repeat_soft_std": 0.08058351278305054, "reward_judge_quality_mean": 0.5562499761581421, "reward_judge_quality_std": 0.2249404937028885, "reward_total_composite_mean": 0.5727308988571167, "reward_total_composite_std": 0.20136988162994385} {"timestamp_utc": "2026-04-13T09:04:57Z", "mode": "train", "global_step": 695, "epoch": 0.06981416373681568, "loss": -0.0441, "grad_norm": 17.265445709228516, "learning_rate": 7.896969696969698e-06, "num_tokens": 1225945.0, "completions/mean_length": 36.375, "completions/min_length": 29.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.375, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.42902883887290955, "rewards/meter/std": 0.45850393176078796, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9710671901702881, "rewards/repeat_soft/std": 0.03647122159600258, "rewards/judge_quality/mean": 0.7325000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.506554901599884, "rewards/total_composite/std": 0.3096925914287567, "reward": 0.506554901599884, "reward_std": 0.3096925616264343, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21644631028175354, "sampling/sampling_logp_difference/max": 1.8106884956359863, "sampling/importance_sampling_ratio/min": 0.1635415107011795, "sampling/importance_sampling_ratio/mean": 1.0212366580963135, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5601515024900436, "clip_ratio/low_mean": 0.12057080771774054, "clip_ratio/low_min": 0.12057080771774054, "clip_ratio/high_mean": 0.07483388856053352, "clip_ratio/high_max": 0.07483388856053352, "clip_ratio/region_mean": 0.19540469627827406, "reward_total_mean": 0.506554901599884, "reward_meter_mean": 0.42902883887290955, "reward_meter_std": 0.45850393176078796, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9710671901702881, "reward_repeat_soft_std": 0.03647122159600258, "reward_judge_quality_mean": 0.7325000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.506554901599884, "reward_total_composite_std": 0.3096925914287567} {"timestamp_utc": "2026-04-13T09:05:04Z", "mode": "train", "global_step": 696, "epoch": 0.06991461577096936, "loss": 0.0479, "grad_norm": 11.10001277923584, "learning_rate": 7.893939393939395e-06, "num_tokens": 1228150.0, "completions/mean_length": 80.625, "completions/min_length": 77.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.625, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.3874858617782593, "rewards/meter/std": 0.2686140239238739, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.8899877071380615, "rewards/repeat_soft/std": 0.0854894369840622, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.3540748953819275, "rewards/total_composite/std": 0.25735554099082947, "reward": 0.3540748953819275, "reward_std": 0.25735554099082947, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1570495069026947, "sampling/sampling_logp_difference/max": 1.967787742614746, "sampling/importance_sampling_ratio/min": 0.13976570963859558, "sampling/importance_sampling_ratio/mean": 1.0180790424346924, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1517724245786667, "clip_ratio/low_mean": 0.040153549052774906, "clip_ratio/low_min": 0.040153549052774906, "clip_ratio/high_mean": 0.08434855472296476, "clip_ratio/high_max": 0.08434855472296476, "clip_ratio/region_mean": 0.12450210377573967, "reward_total_mean": 0.3540748953819275, "reward_meter_mean": 0.3874858617782593, "reward_meter_std": 0.2686140239238739, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.8899877071380615, "reward_repeat_soft_std": 0.0854894369840622, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.3540748953819275, "reward_total_composite_std": 0.25735554099082947} {"timestamp_utc": "2026-04-13T09:05:11Z", "mode": "train", "global_step": 697, "epoch": 0.07001506780512305, "loss": 0.0636, "grad_norm": 16.595430374145508, "learning_rate": 7.89090909090909e-06, "num_tokens": 1229771.0, "completions/mean_length": 38.625, "completions/min_length": 33.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.625, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.8890440464019775, "rewards/meter/std": 0.14752990007400513, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9806209206581116, "rewards/repeat_soft/std": 0.023214057087898254, "rewards/judge_quality/mean": 0.6112500429153442, "rewards/judge_quality/std": 0.25587037205696106, "rewards/total_composite/mean": 0.6869492530822754, "rewards/total_composite/std": 0.12026029080152512, "reward": 0.6869492530822754, "reward_std": 0.12026028335094452, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15296423435211182, "sampling/sampling_logp_difference/max": 1.3662142753601074, "sampling/importance_sampling_ratio/min": 0.2687130868434906, "sampling/importance_sampling_ratio/mean": 1.0238951444625854, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1991469413042068, "clip_ratio/low_mean": 0.09507615398615599, "clip_ratio/low_min": 0.09507615398615599, "clip_ratio/high_mean": 0.060897436924278736, "clip_ratio/high_max": 0.060897436924278736, "clip_ratio/region_mean": 0.15597359091043472, "reward_total_mean": 0.6869492530822754, "reward_meter_mean": 0.8890440464019775, "reward_meter_std": 0.14752990007400513, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9806209206581116, "reward_repeat_soft_std": 0.023214057087898254, "reward_judge_quality_mean": 0.6112500429153442, "reward_judge_quality_std": 0.25587037205696106, "reward_total_composite_mean": 0.6869492530822754, "reward_total_composite_std": 0.12026029080152512} {"timestamp_utc": "2026-04-13T09:05:23Z", "mode": "train", "global_step": 698, "epoch": 0.07011551983927675, "loss": -0.1319, "grad_norm": 2.9320316314697266, "learning_rate": 7.88787878787879e-06, "num_tokens": 1231756.0, "completions/mean_length": 114.125, "completions/min_length": 46.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 57.28571701049805, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.5878835916519165, "rewards/meter/std": 0.27737826108932495, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.880814790725708, "rewards/repeat_soft/std": 0.09316648542881012, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.4399494528770447, "rewards/total_composite/std": 0.19185154139995575, "reward": 0.4399494528770447, "reward_std": 0.19185155630111694, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15781275928020477, "sampling/sampling_logp_difference/max": 1.5711240768432617, "sampling/importance_sampling_ratio/min": 0.20781145989894867, "sampling/importance_sampling_ratio/mean": 1.0406392812728882, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0126184672117233, "clip_ratio/low_mean": 0.01923076994717121, "clip_ratio/low_min": 0.01923076994717121, "clip_ratio/high_mean": 0.09851447865366936, "clip_ratio/high_max": 0.09851447865366936, "clip_ratio/region_mean": 0.11774524860084057, "reward_total_mean": 0.4399494528770447, "reward_meter_mean": 0.5878835916519165, "reward_meter_std": 0.27737826108932495, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.880814790725708, "reward_repeat_soft_std": 0.09316648542881012, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.4399494528770447, "reward_total_composite_std": 0.19185154139995575} {"timestamp_utc": "2026-04-13T09:05:30Z", "mode": "train", "global_step": 699, "epoch": 0.07021597187343044, "loss": -0.0128, "grad_norm": 9.37679386138916, "learning_rate": 7.884848484848485e-06, "num_tokens": 1234569.0, "completions/mean_length": 116.625, "completions/min_length": 108.0, "completions/max_length": 126.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 116.625, "completions/min_terminated_length": 108.0, "completions/max_terminated_length": 126.0, "rewards/meter/mean": 0.43015190958976746, "rewards/meter/std": 0.3419077694416046, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8563051819801331, "rewards/repeat_soft/std": 0.1234939694404602, "rewards/judge_quality/mean": 0.39375001192092896, "rewards/judge_quality/std": 0.15638209879398346, "rewards/total_composite/mean": 0.4412240982055664, "rewards/total_composite/std": 0.1003093272447586, "reward": 0.4412240982055664, "reward_std": 0.10030930489301682, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13429109752178192, "sampling/sampling_logp_difference/max": 2.4016456604003906, "sampling/importance_sampling_ratio/min": 0.09056878834962845, "sampling/importance_sampling_ratio/mean": 1.0258796215057373, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9660872519016266, "clip_ratio/low_mean": 0.052333916537463665, "clip_ratio/low_min": 0.052333916537463665, "clip_ratio/high_mean": 0.06842698622494936, "clip_ratio/high_max": 0.06842698622494936, "clip_ratio/region_mean": 0.12076090276241302, "reward_total_mean": 0.4412240982055664, "reward_meter_mean": 0.43015190958976746, "reward_meter_std": 0.3419077694416046, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8563051819801331, "reward_repeat_soft_std": 0.1234939694404602, "reward_judge_quality_mean": 0.39375001192092896, "reward_judge_quality_std": 0.15638209879398346, "reward_total_composite_mean": 0.4412240982055664, "reward_total_composite_std": 0.1003093272447586} {"timestamp_utc": "2026-04-13T09:05:37Z", "mode": "train", "global_step": 700, "epoch": 0.07031642390758412, "loss": 0.0249, "grad_norm": 10.956107139587402, "learning_rate": 7.881818181818182e-06, "num_tokens": 1236389.0, "completions/mean_length": 56.5, "completions/min_length": 51.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.5, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.699101448059082, "rewards/meter/std": 0.26901111006736755, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9079020023345947, "rewards/repeat_soft/std": 0.044114850461483, "rewards/judge_quality/mean": 0.8075000047683716, "rewards/judge_quality/std": 0.18077215552330017, "rewards/total_composite/mean": 0.6893068552017212, "rewards/total_composite/std": 0.14247862994670868, "reward": 0.6893068552017212, "reward_std": 0.14247861504554749, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16456753015518188, "sampling/sampling_logp_difference/max": 1.9089515209197998, "sampling/importance_sampling_ratio/min": 0.1482357233762741, "sampling/importance_sampling_ratio/mean": 0.9832058548927307, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0508423298597336, "clip_ratio/low_mean": 0.0792367160320282, "clip_ratio/low_min": 0.0792367160320282, "clip_ratio/high_mean": 0.07942836545407772, "clip_ratio/high_max": 0.07942836545407772, "clip_ratio/region_mean": 0.15866508148610592, "reward_total_mean": 0.6893068552017212, "reward_meter_mean": 0.699101448059082, "reward_meter_std": 0.26901111006736755, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9079020023345947, "reward_repeat_soft_std": 0.044114850461483, "reward_judge_quality_mean": 0.8075000047683716, "reward_judge_quality_std": 0.18077215552330017, "reward_total_composite_mean": 0.6893068552017212, "reward_total_composite_std": 0.14247862994670868} {"timestamp_utc": "2026-04-13T09:06:24Z", "mode": "eval", "global_step": 700, "epoch": 0.07031642390758412, "eval_loss": NaN, "eval_runtime": 47.094, "eval_samples_per_second": 1.699, "eval_steps_per_second": 0.212, "eval_num_tokens": 1236389.0, "eval_completions/mean_length": 63.2375, "eval_completions/min_length": 26.6, "eval_completions/max_length": 140.4, "eval_completions/clipped_ratio": 0.0125, "eval_completions/mean_terminated_length": 57.42678604125977, "eval_completions/min_terminated_length": 26.6, "eval_completions/max_terminated_length": 99.0, "eval_rewards/meter/mean": 0.562597519159317, "eval_rewards/meter/std": 0.3473539650440216, "eval_rewards/count_adherence/mean": 0.9895833313465119, "eval_rewards/count_adherence/std": 0.029462784901261328, "eval_rewards/hard_gate/mean": 0.9625, "eval_rewards/hard_gate/std": 0.10606601536273956, "eval_rewards/repeat_soft/mean": 0.875518399477005, "eval_rewards/repeat_soft/std": 0.11491637155413628, "eval_rewards/judge_quality/mean": 0.5047500103712081, "eval_rewards/judge_quality/std": 0.20631255060434342, "eval_rewards/total_composite/mean": 0.4972403317689896, "eval_rewards/total_composite/std": 0.1763676755130291, "eval_reward": 0.4972403317689896, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.0910422645509243, "eval_sampling/sampling_logp_difference/max": 1.0577627658843993, "eval_sampling/importance_sampling_ratio/min": 0.353044393658638, "eval_sampling/importance_sampling_ratio/mean": 1.0241777658462525, "eval_sampling/importance_sampling_ratio/max": 1.5206967353820802, "eval_entropy": 1.0876900553703308, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.4972403317689896, "eval_reward_meter_mean": 0.562597519159317, "eval_reward_meter_std": 0.3473539650440216, "eval_reward_count_adherence_mean": 0.9895833313465119, "eval_reward_count_adherence_std": 0.029462784901261328, "eval_reward_hard_gate_mean": 0.9625, "eval_reward_hard_gate_std": 0.10606601536273956, "eval_reward_repeat_soft_mean": 0.875518399477005, "eval_reward_repeat_soft_std": 0.11491637155413628, "eval_reward_judge_quality_mean": 0.5047500103712081, "eval_reward_judge_quality_std": 0.20631255060434342, "eval_reward_total_composite_mean": 0.4972403317689896, "eval_reward_total_composite_std": 0.1763676755130291} {"timestamp_utc": "2026-04-13T09:06:33Z", "mode": "train", "global_step": 701, "epoch": 0.07041687594173782, "loss": 0.0305, "grad_norm": 16.376840591430664, "learning_rate": 7.87878787878788e-06, "num_tokens": 1237978.0, "completions/mean_length": 32.625, "completions/min_length": 29.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.625, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.48679500818252563, "rewards/meter/std": 0.3600373864173889, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8722328543663025, "rewards/repeat_soft/std": 0.07642845809459686, "rewards/judge_quality/mean": 0.48624998331069946, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.47136837244033813, "rewards/total_composite/std": 0.09856859594583511, "reward": 0.47136837244033813, "reward_std": 0.09856861084699631, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11635728180408478, "sampling/sampling_logp_difference/max": 1.0491828918457031, "sampling/importance_sampling_ratio/min": 0.3502238094806671, "sampling/importance_sampling_ratio/mean": 1.0372222661972046, "sampling/importance_sampling_ratio/max": 1.8948490619659424, "entropy": 0.9957318902015686, "clip_ratio/low_mean": 0.06924716010689735, "clip_ratio/low_min": 0.06924716010689735, "clip_ratio/high_mean": 0.05016236566007137, "clip_ratio/high_max": 0.05016236566007137, "clip_ratio/region_mean": 0.11940952576696873, "reward_total_mean": 0.47136837244033813, "reward_meter_mean": 0.48679500818252563, "reward_meter_std": 0.3600373864173889, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8722328543663025, "reward_repeat_soft_std": 0.07642845809459686, "reward_judge_quality_mean": 0.48624998331069946, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.47136837244033813, "reward_total_composite_std": 0.09856859594583511} {"timestamp_utc": "2026-04-13T09:06:40Z", "mode": "train", "global_step": 702, "epoch": 0.07051732797589151, "loss": -0.0126, "grad_norm": 13.545258522033691, "learning_rate": 7.875757575757577e-06, "num_tokens": 1240039.0, "completions/mean_length": 67.625, "completions/min_length": 61.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.625, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.12041516602039337, "rewards/meter/std": 0.11534888297319412, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8962299227714539, "rewards/repeat_soft/std": 0.11184307932853699, "rewards/judge_quality/mean": 0.5325000286102295, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.38222992420196533, "rewards/total_composite/std": 0.06179346889257431, "reward": 0.38222992420196533, "reward_std": 0.06179346516728401, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1679002195596695, "sampling/sampling_logp_difference/max": 1.4634904861450195, "sampling/importance_sampling_ratio/min": 0.23142707347869873, "sampling/importance_sampling_ratio/mean": 1.0432548522949219, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4721038043498993, "clip_ratio/low_mean": 0.05922293849289417, "clip_ratio/low_min": 0.05922293849289417, "clip_ratio/high_mean": 0.06900245696306229, "clip_ratio/high_max": 0.06900245696306229, "clip_ratio/region_mean": 0.12822539545595646, "reward_total_mean": 0.38222992420196533, "reward_meter_mean": 0.12041516602039337, "reward_meter_std": 0.11534888297319412, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8962299227714539, "reward_repeat_soft_std": 0.11184307932853699, "reward_judge_quality_mean": 0.5325000286102295, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.38222992420196533, "reward_total_composite_std": 0.06179346889257431} {"timestamp_utc": "2026-04-13T09:06:52Z", "mode": "train", "global_step": 703, "epoch": 0.0706177800100452, "loss": -0.1726, "grad_norm": 2.6292431354522705, "learning_rate": 7.872727272727273e-06, "num_tokens": 1242190.0, "completions/mean_length": 140.875, "completions/min_length": 74.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 87.85714721679688, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.7005671262741089, "rewards/meter/std": 0.3341540992259979, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8083329200744629, "rewards/repeat_soft/std": 0.10700494050979614, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.15733835101127625, "rewards/total_composite/mean": 0.4865454435348511, "rewards/total_composite/std": 0.20896413922309875, "reward": 0.4865454435348511, "reward_std": 0.20896413922309875, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1460670679807663, "sampling/sampling_logp_difference/max": 1.56398344039917, "sampling/importance_sampling_ratio/min": 0.2152041792869568, "sampling/importance_sampling_ratio/mean": 1.011475920677185, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.854569286108017, "clip_ratio/low_mean": 0.03128632344305515, "clip_ratio/low_min": 0.03128632344305515, "clip_ratio/high_mean": 0.07816854817792773, "clip_ratio/high_max": 0.07816854817792773, "clip_ratio/region_mean": 0.10945487162098289, "reward_total_mean": 0.4865454435348511, "reward_meter_mean": 0.7005671262741089, "reward_meter_std": 0.3341540992259979, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8083329200744629, "reward_repeat_soft_std": 0.10700494050979614, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.15733835101127625, "reward_total_composite_mean": 0.4865454435348511, "reward_total_composite_std": 0.20896413922309875} {"timestamp_utc": "2026-04-13T09:06:58Z", "mode": "train", "global_step": 704, "epoch": 0.0707182320441989, "loss": 0.0239, "grad_norm": 16.691980361938477, "learning_rate": 7.86969696969697e-06, "num_tokens": 1244265.0, "completions/mean_length": 70.375, "completions/min_length": 62.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.375, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.3741535544395447, "rewards/meter/std": 0.345613956451416, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9200815558433533, "rewards/repeat_soft/std": 0.08679132908582687, "rewards/judge_quality/mean": 0.5199999809265137, "rewards/judge_quality/std": 0.19272483885288239, "rewards/total_composite/mean": 0.37162142992019653, "rewards/total_composite/std": 0.16446079313755035, "reward": 0.37162142992019653, "reward_std": 0.16446080803871155, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20231188833713531, "sampling/sampling_logp_difference/max": 2.399839401245117, "sampling/importance_sampling_ratio/min": 0.09073252975940704, "sampling/importance_sampling_ratio/mean": 0.9932599067687988, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9807969853281975, "clip_ratio/low_mean": 0.014925372786819935, "clip_ratio/low_min": 0.014925372786819935, "clip_ratio/high_mean": 0.18753970600664616, "clip_ratio/high_max": 0.18753970600664616, "clip_ratio/region_mean": 0.2024650787934661, "reward_total_mean": 0.37162142992019653, "reward_meter_mean": 0.3741535544395447, "reward_meter_std": 0.345613956451416, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9200815558433533, "reward_repeat_soft_std": 0.08679132908582687, "reward_judge_quality_mean": 0.5199999809265137, "reward_judge_quality_std": 0.19272483885288239, "reward_total_composite_mean": 0.37162142992019653, "reward_total_composite_std": 0.16446079313755035} {"timestamp_utc": "2026-04-13T09:07:06Z", "mode": "train", "global_step": 705, "epoch": 0.07081868407835258, "loss": 0.0388, "grad_norm": 10.425191879272461, "learning_rate": 7.866666666666667e-06, "num_tokens": 1246456.0, "completions/mean_length": 96.875, "completions/min_length": 88.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.875, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.6700539588928223, "rewards/meter/std": 0.3244590163230896, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9819424152374268, "rewards/repeat_soft/std": 0.01510180626064539, "rewards/judge_quality/mean": 0.5199999809265137, "rewards/judge_quality/std": 0.18516401946544647, "rewards/total_composite/mean": 0.5936654806137085, "rewards/total_composite/std": 0.1849101483821869, "reward": 0.5936654806137085, "reward_std": 0.1849101483821869, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1460622400045395, "sampling/sampling_logp_difference/max": 1.8150241374969482, "sampling/importance_sampling_ratio/min": 0.16283397376537323, "sampling/importance_sampling_ratio/mean": 0.9983146786689758, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7747109085321426, "clip_ratio/low_mean": 0.07223216630518436, "clip_ratio/low_min": 0.07223216630518436, "clip_ratio/high_mean": 0.07081507612019777, "clip_ratio/high_max": 0.07081507612019777, "clip_ratio/region_mean": 0.14304724242538214, "reward_total_mean": 0.5936654806137085, "reward_meter_mean": 0.6700539588928223, "reward_meter_std": 0.3244590163230896, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9819424152374268, "reward_repeat_soft_std": 0.01510180626064539, "reward_judge_quality_mean": 0.5199999809265137, "reward_judge_quality_std": 0.18516401946544647, "reward_total_composite_mean": 0.5936654806137085, "reward_total_composite_std": 0.1849101483821869} {"timestamp_utc": "2026-04-13T09:07:17Z", "mode": "train", "global_step": 706, "epoch": 0.07091913611250628, "loss": -0.1275, "grad_norm": 6.171966075897217, "learning_rate": 7.863636363636364e-06, "num_tokens": 1248227.0, "completions/mean_length": 116.375, "completions/min_length": 46.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 59.857147216796875, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.20292825996875763, "rewards/meter/std": 0.3384077847003937, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9816411733627319, "rewards/repeat_soft/std": 0.02059292607009411, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.23445606231689453, "rewards/total_composite/mean": 0.3845466673374176, "rewards/total_composite/std": 0.2538537383079529, "reward": 0.3845466673374176, "reward_std": 0.2538537383079529, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22096796333789825, "sampling/sampling_logp_difference/max": 1.3180179595947266, "sampling/importance_sampling_ratio/min": 0.267665296792984, "sampling/importance_sampling_ratio/mean": 1.0719823837280273, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.050292491912842, "clip_ratio/low_mean": 0.1373369675129652, "clip_ratio/low_min": 0.1373369675129652, "clip_ratio/high_mean": 0.01909722201526165, "clip_ratio/high_max": 0.01909722201526165, "clip_ratio/region_mean": 0.15643418952822685, "reward_total_mean": 0.3845466673374176, "reward_meter_mean": 0.20292825996875763, "reward_meter_std": 0.3384077847003937, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9816411733627319, "reward_repeat_soft_std": 0.02059292607009411, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.23445606231689453, "reward_total_composite_mean": 0.3845466673374176, "reward_total_composite_std": 0.2538537383079529} {"timestamp_utc": "2026-04-13T09:07:25Z", "mode": "train", "global_step": 707, "epoch": 0.07101958814665997, "loss": -0.015, "grad_norm": 8.014169692993164, "learning_rate": 7.860606060606062e-06, "num_tokens": 1250734.0, "completions/mean_length": 112.375, "completions/min_length": 105.0, "completions/max_length": 124.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.375, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.5072179436683655, "rewards/meter/std": 0.3544645607471466, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.7135148048400879, "rewards/repeat_soft/std": 0.0966494232416153, "rewards/judge_quality/mean": 0.4362499713897705, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.423446387052536, "rewards/total_composite/std": 0.22093810141086578, "reward": 0.423446387052536, "reward_std": 0.2209380865097046, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11801961064338684, "sampling/sampling_logp_difference/max": 1.608119249343872, "sampling/importance_sampling_ratio/min": 0.20026391744613647, "sampling/importance_sampling_ratio/mean": 1.0145412683486938, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8730439320206642, "clip_ratio/low_mean": 0.03968938160687685, "clip_ratio/low_min": 0.03968938160687685, "clip_ratio/high_mean": 0.07308758050203323, "clip_ratio/high_max": 0.07308758050203323, "clip_ratio/region_mean": 0.11277696210891008, "reward_total_mean": 0.423446387052536, "reward_meter_mean": 0.5072179436683655, "reward_meter_std": 0.3544645607471466, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.7135148048400879, "reward_repeat_soft_std": 0.0966494232416153, "reward_judge_quality_mean": 0.4362499713897705, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.423446387052536, "reward_total_composite_std": 0.22093810141086578} {"timestamp_utc": "2026-04-13T09:07:37Z", "mode": "train", "global_step": 708, "epoch": 0.07112004018081367, "loss": -0.1553, "grad_norm": 1.827104926109314, "learning_rate": 7.857575757575759e-06, "num_tokens": 1252520.0, "completions/mean_length": 247.25, "completions/min_length": 83.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 88.4000015258789, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.584979772567749, "rewards/meter/std": 0.2787756323814392, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.31160587072372437, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.7171851396560669, "rewards/repeat_soft/std": 0.09273076057434082, "rewards/judge_quality/mean": 0.23874999582767487, "rewards/judge_quality/std": 0.17141741514205933, "rewards/total_composite/mean": 0.2731322646141052, "rewards/total_composite/std": 0.22805926203727722, "reward": 0.2731322646141052, "reward_std": 0.22805923223495483, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13770842552185059, "sampling/sampling_logp_difference/max": 2.203799247741699, "sampling/importance_sampling_ratio/min": 0.11038298159837723, "sampling/importance_sampling_ratio/mean": 1.0084809064865112, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6015427559614182, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.08695725258439779, "clip_ratio/high_max": 0.08695725258439779, "clip_ratio/region_mean": 0.08695725258439779, "reward_total_mean": 0.2731322646141052, "reward_meter_mean": 0.584979772567749, "reward_meter_std": 0.2787756323814392, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.31160587072372437, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.7171851396560669, "reward_repeat_soft_std": 0.09273076057434082, "reward_judge_quality_mean": 0.23874999582767487, "reward_judge_quality_std": 0.17141741514205933, "reward_total_composite_mean": 0.2731322646141052, "reward_total_composite_std": 0.22805926203727722} {"timestamp_utc": "2026-04-13T09:07:44Z", "mode": "train", "global_step": 709, "epoch": 0.07122049221496736, "loss": 0.0795, "grad_norm": 15.316978454589844, "learning_rate": 7.854545454545454e-06, "num_tokens": 1253990.0, "completions/mean_length": 33.75, "completions/min_length": 29.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.75, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.4370405077934265, "rewards/meter/std": 0.31697505712509155, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9854952096939087, "rewards/repeat_soft/std": 0.012407306581735611, "rewards/judge_quality/mean": 0.5600000023841858, "rewards/judge_quality/std": 0.22258225083351135, "rewards/total_composite/mean": 0.5258222222328186, "rewards/total_composite/std": 0.16406190395355225, "reward": 0.5258222222328186, "reward_std": 0.16406190395355225, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1508551388978958, "sampling/sampling_logp_difference/max": 1.278902530670166, "sampling/importance_sampling_ratio/min": 0.278342604637146, "sampling/importance_sampling_ratio/mean": 1.0059592723846436, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0151208117604256, "clip_ratio/low_mean": 0.06558240670710802, "clip_ratio/low_min": 0.06558240670710802, "clip_ratio/high_mean": 0.06477042380720377, "clip_ratio/high_max": 0.06477042380720377, "clip_ratio/region_mean": 0.1303528305143118, "reward_total_mean": 0.5258222222328186, "reward_meter_mean": 0.4370405077934265, "reward_meter_std": 0.31697505712509155, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9854952096939087, "reward_repeat_soft_std": 0.012407306581735611, "reward_judge_quality_mean": 0.5600000023841858, "reward_judge_quality_std": 0.22258225083351135, "reward_total_composite_mean": 0.5258222222328186, "reward_total_composite_std": 0.16406190395355225} {"timestamp_utc": "2026-04-13T09:07:51Z", "mode": "train", "global_step": 710, "epoch": 0.07132094424912104, "loss": 0.0538, "grad_norm": 13.484710693359375, "learning_rate": 7.851515151515152e-06, "num_tokens": 1255632.0, "completions/mean_length": 42.25, "completions/min_length": 36.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.25, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.8817436099052429, "rewards/meter/std": 0.178593248128891, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.799583375453949, "rewards/repeat_soft/std": 0.14987260103225708, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.5841915011405945, "rewards/total_composite/std": 0.09952611476182938, "reward": 0.5841915011405945, "reward_std": 0.09952611476182938, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1483382135629654, "sampling/sampling_logp_difference/max": 1.8885650634765625, "sampling/importance_sampling_ratio/min": 0.15128874778747559, "sampling/importance_sampling_ratio/mean": 1.0168503522872925, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.999920554459095, "clip_ratio/low_mean": 0.08082265174016356, "clip_ratio/low_min": 0.08082265174016356, "clip_ratio/high_mean": 0.04457815829664469, "clip_ratio/high_max": 0.04457815829664469, "clip_ratio/region_mean": 0.12540081003680825, "reward_total_mean": 0.5841915011405945, "reward_meter_mean": 0.8817436099052429, "reward_meter_std": 0.178593248128891, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.799583375453949, "reward_repeat_soft_std": 0.14987260103225708, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.5841915011405945, "reward_total_composite_std": 0.09952611476182938} {"timestamp_utc": "2026-04-13T09:07:57Z", "mode": "train", "global_step": 711, "epoch": 0.07142139628327474, "loss": 0.0095, "grad_norm": 16.092296600341797, "learning_rate": 7.848484848484849e-06, "num_tokens": 1257169.0, "completions/mean_length": 31.125, "completions/min_length": 30.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.125, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.6596676111221313, "rewards/meter/std": 0.38747116923332214, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9813710451126099, "rewards/repeat_soft/std": 0.017808331176638603, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.6435909271240234, "rewards/total_composite/std": 0.24125070869922638, "reward": 0.6435909271240234, "reward_std": 0.24125069379806519, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12443126738071442, "sampling/sampling_logp_difference/max": 2.83040452003479, "sampling/importance_sampling_ratio/min": 0.058988988399505615, "sampling/importance_sampling_ratio/mean": 1.0007023811340332, "sampling/importance_sampling_ratio/max": 1.543837308883667, "entropy": 0.8394577726721764, "clip_ratio/low_mean": 0.08185483934357762, "clip_ratio/low_min": 0.08185483934357762, "clip_ratio/high_mean": 0.06256670970469713, "clip_ratio/high_max": 0.06256670970469713, "clip_ratio/region_mean": 0.14442154904827476, "reward_total_mean": 0.6435909271240234, "reward_meter_mean": 0.6596676111221313, "reward_meter_std": 0.38747116923332214, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9813710451126099, "reward_repeat_soft_std": 0.017808331176638603, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.6435909271240234, "reward_total_composite_std": 0.24125070869922638} {"timestamp_utc": "2026-04-13T09:08:04Z", "mode": "train", "global_step": 712, "epoch": 0.07152184831742843, "loss": 0.0357, "grad_norm": 18.334585189819336, "learning_rate": 7.845454545454546e-06, "num_tokens": 1258632.0, "completions/mean_length": 34.875, "completions/min_length": 30.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.875, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.14928486943244934, "rewards/meter/std": 0.14895722270011902, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9914079308509827, "rewards/repeat_soft/std": 0.010679380968213081, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.39553284645080566, "rewards/total_composite/std": 0.044622622430324554, "reward": 0.39553284645080566, "reward_std": 0.04462261125445366, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16986387968063354, "sampling/sampling_logp_difference/max": 1.7969157695770264, "sampling/importance_sampling_ratio/min": 0.1658094972372055, "sampling/importance_sampling_ratio/mean": 1.0382651090621948, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4820304661989212, "clip_ratio/low_mean": 0.08604302210733294, "clip_ratio/low_min": 0.08604302210733294, "clip_ratio/high_mean": 0.06095559988170862, "clip_ratio/high_max": 0.06095559988170862, "clip_ratio/region_mean": 0.14699862198904157, "reward_total_mean": 0.39553284645080566, "reward_meter_mean": 0.14928486943244934, "reward_meter_std": 0.14895722270011902, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9914079308509827, "reward_repeat_soft_std": 0.010679380968213081, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.39553284645080566, "reward_total_composite_std": 0.044622622430324554} {"timestamp_utc": "2026-04-13T09:08:11Z", "mode": "train", "global_step": 713, "epoch": 0.07162230035158212, "loss": 0.0261, "grad_norm": 17.9517879486084, "learning_rate": 7.842424242424243e-06, "num_tokens": 1260195.0, "completions/mean_length": 36.375, "completions/min_length": 29.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.375, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.8564504981040955, "rewards/meter/std": 0.21740178763866425, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9412529468536377, "rewards/repeat_soft/std": 0.07152710855007172, "rewards/judge_quality/mean": 0.5649999976158142, "rewards/judge_quality/std": 0.194054514169693, "rewards/total_composite/mean": 0.6544015407562256, "rewards/total_composite/std": 0.1400221437215805, "reward": 0.6544015407562256, "reward_std": 0.1400221586227417, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18278828263282776, "sampling/sampling_logp_difference/max": 1.9740679264068604, "sampling/importance_sampling_ratio/min": 0.13889069855213165, "sampling/importance_sampling_ratio/mean": 0.9994227886199951, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0823270753026009, "clip_ratio/low_mean": 0.07796026673167944, "clip_ratio/low_min": 0.07796026673167944, "clip_ratio/high_mean": 0.0432076808065176, "clip_ratio/high_max": 0.0432076808065176, "clip_ratio/region_mean": 0.12116794753819704, "reward_total_mean": 0.6544015407562256, "reward_meter_mean": 0.8564504981040955, "reward_meter_std": 0.21740178763866425, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9412529468536377, "reward_repeat_soft_std": 0.07152710855007172, "reward_judge_quality_mean": 0.5649999976158142, "reward_judge_quality_std": 0.194054514169693, "reward_total_composite_mean": 0.6544015407562256, "reward_total_composite_std": 0.1400221437215805} {"timestamp_utc": "2026-04-13T09:08:18Z", "mode": "train", "global_step": 714, "epoch": 0.0717227523857358, "loss": 0.0862, "grad_norm": 20.394691467285156, "learning_rate": 7.83939393939394e-06, "num_tokens": 1261660.0, "completions/mean_length": 23.125, "completions/min_length": 17.0, "completions/max_length": 27.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.125, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.5563315749168396, "rewards/meter/std": 0.46619367599487305, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.926868736743927, "rewards/repeat_soft/std": 0.08202483505010605, "rewards/judge_quality/mean": 0.44624999165534973, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.5017377138137817, "rewards/total_composite/std": 0.14172938466072083, "reward": 0.5017377138137817, "reward_std": 0.14172936975955963, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1598486304283142, "sampling/sampling_logp_difference/max": 1.5706138610839844, "sampling/importance_sampling_ratio/min": 0.2079174965620041, "sampling/importance_sampling_ratio/mean": 1.0144904851913452, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1529867574572563, "clip_ratio/low_mean": 0.030833333730697632, "clip_ratio/low_min": 0.030833333730697632, "clip_ratio/high_mean": 0.10804452188313007, "clip_ratio/high_max": 0.10804452188313007, "clip_ratio/region_mean": 0.1388778556138277, "reward_total_mean": 0.5017377138137817, "reward_meter_mean": 0.5563315749168396, "reward_meter_std": 0.46619367599487305, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.926868736743927, "reward_repeat_soft_std": 0.08202483505010605, "reward_judge_quality_mean": 0.44624999165534973, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.5017377138137817, "reward_total_composite_std": 0.14172938466072083} {"timestamp_utc": "2026-04-13T09:08:25Z", "mode": "train", "global_step": 715, "epoch": 0.0718232044198895, "loss": -0.0115, "grad_norm": 12.384317398071289, "learning_rate": 7.836363636363638e-06, "num_tokens": 1263534.0, "completions/mean_length": 70.25, "completions/min_length": 61.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.25, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.37978553771972656, "rewards/meter/std": 0.3818439245223999, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9859769344329834, "rewards/repeat_soft/std": 0.007877059280872345, "rewards/judge_quality/mean": 0.7487500309944153, "rewards/judge_quality/std": 0.0699872374534607, "rewards/total_composite/mean": 0.5291460156440735, "rewards/total_composite/std": 0.18106192350387573, "reward": 0.5291460156440735, "reward_std": 0.18106190860271454, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1295756697654724, "sampling/sampling_logp_difference/max": 1.5767947435379028, "sampling/importance_sampling_ratio/min": 0.20663635432720184, "sampling/importance_sampling_ratio/mean": 1.0221647024154663, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7575226053595543, "clip_ratio/low_mean": 0.07443731464445591, "clip_ratio/low_min": 0.07443731464445591, "clip_ratio/high_mean": 0.046102119609713554, "clip_ratio/high_max": 0.046102119609713554, "clip_ratio/region_mean": 0.12053943425416946, "reward_total_mean": 0.5291460156440735, "reward_meter_mean": 0.37978553771972656, "reward_meter_std": 0.3818439245223999, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9859769344329834, "reward_repeat_soft_std": 0.007877059280872345, "reward_judge_quality_mean": 0.7487500309944153, "reward_judge_quality_std": 0.0699872374534607, "reward_total_composite_mean": 0.5291460156440735, "reward_total_composite_std": 0.18106192350387573} {"timestamp_utc": "2026-04-13T09:08:32Z", "mode": "train", "global_step": 716, "epoch": 0.0719236564540432, "loss": 0.0699, "grad_norm": 12.60446834564209, "learning_rate": 7.833333333333333e-06, "num_tokens": 1265443.0, "completions/mean_length": 62.625, "completions/min_length": 56.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.625, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.2881770730018616, "rewards/meter/std": 0.13510683178901672, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9140480160713196, "rewards/repeat_soft/std": 0.09184758365154266, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.4229186177253723, "rewards/total_composite/std": 0.05308286473155022, "reward": 0.4229186177253723, "reward_std": 0.05308286473155022, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15849167108535767, "sampling/sampling_logp_difference/max": 1.5524301528930664, "sampling/importance_sampling_ratio/min": 0.21173278987407684, "sampling/importance_sampling_ratio/mean": 1.0272974967956543, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2141590863466263, "clip_ratio/low_mean": 0.09201069548726082, "clip_ratio/low_min": 0.09201069548726082, "clip_ratio/high_mean": 0.05910267308354378, "clip_ratio/high_max": 0.05910267308354378, "clip_ratio/region_mean": 0.1511133685708046, "reward_total_mean": 0.4229186177253723, "reward_meter_mean": 0.2881770730018616, "reward_meter_std": 0.13510683178901672, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9140480160713196, "reward_repeat_soft_std": 0.09184758365154266, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.4229186177253723, "reward_total_composite_std": 0.05308286473155022} {"timestamp_utc": "2026-04-13T09:08:44Z", "mode": "train", "global_step": 717, "epoch": 0.07202410848819689, "loss": -0.1462, "grad_norm": 3.343062400817871, "learning_rate": 7.83030303030303e-06, "num_tokens": 1267277.0, "completions/mean_length": 120.25, "completions/min_length": 59.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 64.28572082519531, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.31134355068206787, "rewards/meter/std": 0.24840696156024933, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8691983222961426, "rewards/repeat_soft/std": 0.09458259493112564, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.23445606231689453, "rewards/total_composite/mean": 0.3813011646270752, "rewards/total_composite/std": 0.173007532954216, "reward": 0.3813011646270752, "reward_std": 0.1730075180530548, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13522842526435852, "sampling/sampling_logp_difference/max": 1.5793147087097168, "sampling/importance_sampling_ratio/min": 0.20611628890037537, "sampling/importance_sampling_ratio/mean": 1.0342615842819214, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9020535796880722, "clip_ratio/low_mean": 0.05329301115125418, "clip_ratio/low_min": 0.05329301115125418, "clip_ratio/high_mean": 0.07759564556181431, "clip_ratio/high_max": 0.07759564556181431, "clip_ratio/region_mean": 0.13088865671306849, "reward_total_mean": 0.3813011646270752, "reward_meter_mean": 0.31134355068206787, "reward_meter_std": 0.24840696156024933, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8691983222961426, "reward_repeat_soft_std": 0.09458259493112564, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.23445606231689453, "reward_total_composite_mean": 0.3813011646270752, "reward_total_composite_std": 0.173007532954216} {"timestamp_utc": "2026-04-13T09:08:51Z", "mode": "train", "global_step": 718, "epoch": 0.07212456052235058, "loss": 0.0157, "grad_norm": 18.708446502685547, "learning_rate": 7.827272727272728e-06, "num_tokens": 1268898.0, "completions/mean_length": 41.625, "completions/min_length": 40.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.625, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.7582693099975586, "rewards/meter/std": 0.3461999297142029, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9715173244476318, "rewards/repeat_soft/std": 0.04452996328473091, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.5789026021957397, "rewards/total_composite/std": 0.09932930022478104, "reward": 0.5789026021957397, "reward_std": 0.09932930022478104, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12216933816671371, "sampling/sampling_logp_difference/max": 1.7323646545410156, "sampling/importance_sampling_ratio/min": 0.17686569690704346, "sampling/importance_sampling_ratio/mean": 1.0042716264724731, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8733915016055107, "clip_ratio/low_mean": 0.03689024318009615, "clip_ratio/low_min": 0.03689024318009615, "clip_ratio/high_mean": 0.10406699776649475, "clip_ratio/high_max": 0.10406699776649475, "clip_ratio/region_mean": 0.1409572409465909, "reward_total_mean": 0.5789026021957397, "reward_meter_mean": 0.7582693099975586, "reward_meter_std": 0.3461999297142029, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9715173244476318, "reward_repeat_soft_std": 0.04452996328473091, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.5789026021957397, "reward_total_composite_std": 0.09932930022478104} {"timestamp_utc": "2026-04-13T09:08:57Z", "mode": "train", "global_step": 719, "epoch": 0.07222501255650426, "loss": 0.0427, "grad_norm": 18.660600662231445, "learning_rate": 7.824242424242425e-06, "num_tokens": 1270425.0, "completions/mean_length": 32.875, "completions/min_length": 28.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.875, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.7567814588546753, "rewards/meter/std": 0.3158141076564789, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9840539693832397, "rewards/repeat_soft/std": 0.007520779967308044, "rewards/judge_quality/mean": 0.48874998092651367, "rewards/judge_quality/std": 0.28048110008239746, "rewards/total_composite/mean": 0.5574514269828796, "rewards/total_composite/std": 0.13825681805610657, "reward": 0.5574514269828796, "reward_std": 0.13825680315494537, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15731342136859894, "sampling/sampling_logp_difference/max": 1.4657902717590332, "sampling/importance_sampling_ratio/min": 0.23089544475078583, "sampling/importance_sampling_ratio/mean": 1.0008772611618042, "sampling/importance_sampling_ratio/max": 1.9630663394927979, "entropy": 0.9993829131126404, "clip_ratio/low_mean": 0.06591235660016537, "clip_ratio/low_min": 0.06591235660016537, "clip_ratio/high_mean": 0.05210291314870119, "clip_ratio/high_max": 0.05210291314870119, "clip_ratio/region_mean": 0.11801526974886656, "reward_total_mean": 0.5574514269828796, "reward_meter_mean": 0.7567814588546753, "reward_meter_std": 0.3158141076564789, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9840539693832397, "reward_repeat_soft_std": 0.007520779967308044, "reward_judge_quality_mean": 0.48874998092651367, "reward_judge_quality_std": 0.28048110008239746, "reward_total_composite_mean": 0.5574514269828796, "reward_total_composite_std": 0.13825681805610657} {"timestamp_utc": "2026-04-13T09:09:03Z", "mode": "train", "global_step": 720, "epoch": 0.07232546459065796, "loss": 0.0734, "grad_norm": 23.673458099365234, "learning_rate": 7.821212121212122e-06, "num_tokens": 1271911.0, "completions/mean_length": 19.75, "completions/min_length": 14.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 19.75, "completions/min_terminated_length": 14.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.8487032651901245, "rewards/meter/std": 0.29530757665634155, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9546234011650085, "rewards/repeat_soft/std": 0.022278277203440666, "rewards/judge_quality/mean": 0.3087500035762787, "rewards/judge_quality/std": 0.1470119059085846, "rewards/total_composite/mean": 0.5278379321098328, "rewards/total_composite/std": 0.11549441516399384, "reward": 0.5278379321098328, "reward_std": 0.11549440771341324, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17154358327388763, "sampling/sampling_logp_difference/max": 1.6976003646850586, "sampling/importance_sampling_ratio/min": 0.1831224262714386, "sampling/importance_sampling_ratio/mean": 0.9992610216140747, "sampling/importance_sampling_ratio/max": 1.7448310852050781, "entropy": 1.1536924913525581, "clip_ratio/low_mean": 0.09748476650565863, "clip_ratio/low_min": 0.09748476650565863, "clip_ratio/high_mean": 0.09762183204293251, "clip_ratio/high_max": 0.09762183204293251, "clip_ratio/region_mean": 0.19510659854859114, "reward_total_mean": 0.5278379321098328, "reward_meter_mean": 0.8487032651901245, "reward_meter_std": 0.29530757665634155, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9546234011650085, "reward_repeat_soft_std": 0.022278277203440666, "reward_judge_quality_mean": 0.3087500035762787, "reward_judge_quality_std": 0.1470119059085846, "reward_total_composite_mean": 0.5278379321098328, "reward_total_composite_std": 0.11549441516399384} {"timestamp_utc": "2026-04-13T09:09:10Z", "mode": "train", "global_step": 721, "epoch": 0.07242591662481165, "loss": 0.0194, "grad_norm": 7.824896812438965, "learning_rate": 7.81818181818182e-06, "num_tokens": 1274152.0, "completions/mean_length": 104.125, "completions/min_length": 87.0, "completions/max_length": 124.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 104.125, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.4792739450931549, "rewards/meter/std": 0.3051510751247406, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6932368874549866, "rewards/repeat_soft/std": 0.16597597301006317, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.4246557354927063, "rewards/total_composite/std": 0.08762908726930618, "reward": 0.4246557354927063, "reward_std": 0.08762907981872559, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11447266489267349, "sampling/sampling_logp_difference/max": 2.2866077423095703, "sampling/importance_sampling_ratio/min": 0.10161057114601135, "sampling/importance_sampling_ratio/mean": 0.9976920485496521, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8533372655510902, "clip_ratio/low_mean": 0.04566267365589738, "clip_ratio/low_min": 0.04566267365589738, "clip_ratio/high_mean": 0.06220975797623396, "clip_ratio/high_max": 0.06220975797623396, "clip_ratio/region_mean": 0.10787243163213134, "reward_total_mean": 0.4246557354927063, "reward_meter_mean": 0.4792739450931549, "reward_meter_std": 0.3051510751247406, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6932368874549866, "reward_repeat_soft_std": 0.16597597301006317, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.4246557354927063, "reward_total_composite_std": 0.08762908726930618} {"timestamp_utc": "2026-04-13T09:09:16Z", "mode": "train", "global_step": 722, "epoch": 0.07252636865896535, "loss": 0.0906, "grad_norm": 15.575541496276855, "learning_rate": 7.815151515151515e-06, "num_tokens": 1276024.0, "completions/mean_length": 57.0, "completions/min_length": 47.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.0, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.655722439289093, "rewards/meter/std": 0.35337594151496887, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9352512359619141, "rewards/repeat_soft/std": 0.04997103288769722, "rewards/judge_quality/mean": 0.5649999976158142, "rewards/judge_quality/std": 0.20057062804698944, "rewards/total_composite/mean": 0.6082093715667725, "rewards/total_composite/std": 0.20887351036071777, "reward": 0.6082093715667725, "reward_std": 0.20887349545955658, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1639634221792221, "sampling/sampling_logp_difference/max": 1.4926071166992188, "sampling/importance_sampling_ratio/min": 0.22478584945201874, "sampling/importance_sampling_ratio/mean": 1.0217629671096802, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2073974460363388, "clip_ratio/low_mean": 0.06932212691754103, "clip_ratio/low_min": 0.06932212691754103, "clip_ratio/high_mean": 0.07310424745082855, "clip_ratio/high_max": 0.07310424745082855, "clip_ratio/region_mean": 0.14242637436836958, "reward_total_mean": 0.6082093715667725, "reward_meter_mean": 0.655722439289093, "reward_meter_std": 0.35337594151496887, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9352512359619141, "reward_repeat_soft_std": 0.04997103288769722, "reward_judge_quality_mean": 0.5649999976158142, "reward_judge_quality_std": 0.20057062804698944, "reward_total_composite_mean": 0.6082093715667725, "reward_total_composite_std": 0.20887351036071777} {"timestamp_utc": "2026-04-13T09:09:28Z", "mode": "train", "global_step": 723, "epoch": 0.07262682069311903, "loss": -0.0893, "grad_norm": 3.653560161590576, "learning_rate": 7.812121212121213e-06, "num_tokens": 1277560.0, "completions/mean_length": 104.0, "completions/min_length": 35.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 45.71428680419922, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.8652969598770142, "rewards/meter/std": 0.2713530361652374, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9274253845214844, "rewards/repeat_soft/std": 0.04185902699828148, "rewards/judge_quality/mean": 0.518750011920929, "rewards/judge_quality/std": 0.2695730924606323, "rewards/total_composite/mean": 0.5890629291534424, "rewards/total_composite/std": 0.2866470217704773, "reward": 0.5890629291534424, "reward_std": 0.2866469919681549, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15090833604335785, "sampling/sampling_logp_difference/max": 1.5242571830749512, "sampling/importance_sampling_ratio/min": 0.21778278052806854, "sampling/importance_sampling_ratio/mean": 1.0158202648162842, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8286675438284874, "clip_ratio/low_mean": 0.03413120657205582, "clip_ratio/low_min": 0.03413120657205582, "clip_ratio/high_mean": 0.08307747496291995, "clip_ratio/high_max": 0.08307747496291995, "clip_ratio/region_mean": 0.11720868153497577, "reward_total_mean": 0.5890629291534424, "reward_meter_mean": 0.8652969598770142, "reward_meter_std": 0.2713530361652374, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9274253845214844, "reward_repeat_soft_std": 0.04185902699828148, "reward_judge_quality_mean": 0.518750011920929, "reward_judge_quality_std": 0.2695730924606323, "reward_total_composite_mean": 0.5890629291534424, "reward_total_composite_std": 0.2866470217704773} {"timestamp_utc": "2026-04-13T09:09:34Z", "mode": "train", "global_step": 724, "epoch": 0.07272727272727272, "loss": -0.0251, "grad_norm": 24.232315063476562, "learning_rate": 7.80909090909091e-06, "num_tokens": 1279079.0, "completions/mean_length": 18.875, "completions/min_length": 15.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 18.875, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.7440844774246216, "rewards/meter/std": 0.4326940178871155, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9621384143829346, "rewards/repeat_soft/std": 0.0010226722806692123, "rewards/judge_quality/mean": 0.2587500214576721, "rewards/judge_quality/std": 0.15037453174591064, "rewards/total_composite/mean": 0.4419988989830017, "rewards/total_composite/std": 0.08182511478662491, "reward": 0.4419988989830017, "reward_std": 0.08182510733604431, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13539238274097443, "sampling/sampling_logp_difference/max": 1.2402918338775635, "sampling/importance_sampling_ratio/min": 0.28929978609085083, "sampling/importance_sampling_ratio/mean": 1.0327802896499634, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8540509678423405, "clip_ratio/low_mean": 0.07351715723052621, "clip_ratio/low_min": 0.07351715723052621, "clip_ratio/high_mean": 0.01923076994717121, "clip_ratio/high_max": 0.01923076994717121, "clip_ratio/region_mean": 0.09274792717769742, "reward_total_mean": 0.4419988989830017, "reward_meter_mean": 0.7440844774246216, "reward_meter_std": 0.4326940178871155, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9621384143829346, "reward_repeat_soft_std": 0.0010226722806692123, "reward_judge_quality_mean": 0.2587500214576721, "reward_judge_quality_std": 0.15037453174591064, "reward_total_composite_mean": 0.4419988989830017, "reward_total_composite_std": 0.08182511478662491} {"timestamp_utc": "2026-04-13T09:09:40Z", "mode": "train", "global_step": 725, "epoch": 0.07282772476142642, "loss": 0.0342, "grad_norm": 20.793373107910156, "learning_rate": 7.806060606060607e-06, "num_tokens": 1281054.0, "completions/mean_length": 64.875, "completions/min_length": 52.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.875, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.5379571914672852, "rewards/meter/std": 0.27355244755744934, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9909393787384033, "rewards/repeat_soft/std": 0.010423214174807072, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.5543895959854126, "rewards/total_composite/std": 0.08682841807603836, "reward": 0.5543895959854126, "reward_std": 0.08682840317487717, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1904158592224121, "sampling/sampling_logp_difference/max": 2.63891339302063, "sampling/importance_sampling_ratio/min": 0.07143885642290115, "sampling/importance_sampling_ratio/mean": 1.0042673349380493, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7586786299943924, "clip_ratio/low_mean": 0.09301826171576977, "clip_ratio/low_min": 0.09301826171576977, "clip_ratio/high_mean": 0.07457197550684214, "clip_ratio/high_max": 0.07457197550684214, "clip_ratio/region_mean": 0.1675902372226119, "reward_total_mean": 0.5543895959854126, "reward_meter_mean": 0.5379571914672852, "reward_meter_std": 0.27355244755744934, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9909393787384033, "reward_repeat_soft_std": 0.010423214174807072, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.5543895959854126, "reward_total_composite_std": 0.08682841807603836} {"timestamp_utc": "2026-04-13T09:09:48Z", "mode": "train", "global_step": 726, "epoch": 0.07292817679558011, "loss": 0.0355, "grad_norm": 8.718574523925781, "learning_rate": 7.803030303030303e-06, "num_tokens": 1283328.0, "completions/mean_length": 113.25, "completions/min_length": 97.0, "completions/max_length": 131.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 113.25, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.5364246368408203, "rewards/meter/std": 0.3864142596721649, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6865354776382446, "rewards/repeat_soft/std": 0.20342059433460236, "rewards/judge_quality/mean": 0.5199999809265137, "rewards/judge_quality/std": 0.19272483885288239, "rewards/total_composite/mean": 0.4773806929588318, "rewards/total_composite/std": 0.11725659668445587, "reward": 0.4773806929588318, "reward_std": 0.11725659668445587, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13357757031917572, "sampling/sampling_logp_difference/max": 2.550213575363159, "sampling/importance_sampling_ratio/min": 0.07806499302387238, "sampling/importance_sampling_ratio/mean": 1.0096253156661987, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8650649413466454, "clip_ratio/low_mean": 0.06561764236539602, "clip_ratio/low_min": 0.06561764236539602, "clip_ratio/high_mean": 0.059514397755265236, "clip_ratio/high_max": 0.059514397755265236, "clip_ratio/region_mean": 0.12513204012066126, "reward_total_mean": 0.4773806929588318, "reward_meter_mean": 0.5364246368408203, "reward_meter_std": 0.3864142596721649, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6865354776382446, "reward_repeat_soft_std": 0.20342059433460236, "reward_judge_quality_mean": 0.5199999809265137, "reward_judge_quality_std": 0.19272483885288239, "reward_total_composite_mean": 0.4773806929588318, "reward_total_composite_std": 0.11725659668445587} {"timestamp_utc": "2026-04-13T09:09:54Z", "mode": "train", "global_step": 727, "epoch": 0.07302862882973381, "loss": -0.0132, "grad_norm": 15.35572624206543, "learning_rate": 7.800000000000002e-06, "num_tokens": 1285105.0, "completions/mean_length": 42.125, "completions/min_length": 36.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.125, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.835135817527771, "rewards/meter/std": 0.3140829801559448, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9693187475204468, "rewards/repeat_soft/std": 0.043177492916584015, "rewards/judge_quality/mean": 0.8575000166893005, "rewards/judge_quality/std": 0.176776722073555, "rewards/total_composite/mean": 0.8076326847076416, "rewards/total_composite/std": 0.2045147866010666, "reward": 0.8076326847076416, "reward_std": 0.2045147866010666, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18121851980686188, "sampling/sampling_logp_difference/max": 3.835233211517334, "sampling/importance_sampling_ratio/min": 0.021596301347017288, "sampling/importance_sampling_ratio/mean": 1.021759271621704, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3519465550780296, "clip_ratio/low_mean": 0.04869186133146286, "clip_ratio/low_min": 0.04869186133146286, "clip_ratio/high_mean": 0.13332146871834993, "clip_ratio/high_max": 0.13332146871834993, "clip_ratio/region_mean": 0.1820133300498128, "reward_total_mean": 0.8076326847076416, "reward_meter_mean": 0.835135817527771, "reward_meter_std": 0.3140829801559448, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9693187475204468, "reward_repeat_soft_std": 0.043177492916584015, "reward_judge_quality_mean": 0.8575000166893005, "reward_judge_quality_std": 0.176776722073555, "reward_total_composite_mean": 0.8076326847076416, "reward_total_composite_std": 0.2045147866010666} {"timestamp_utc": "2026-04-13T09:10:00Z", "mode": "train", "global_step": 728, "epoch": 0.07312908086388749, "loss": 0.0498, "grad_norm": 15.05860710144043, "learning_rate": 7.796969696969697e-06, "num_tokens": 1286799.0, "completions/mean_length": 49.75, "completions/min_length": 44.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.75, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.4592335820198059, "rewards/meter/std": 0.3384995758533478, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9714906215667725, "rewards/repeat_soft/std": 0.022578958421945572, "rewards/judge_quality/mean": 0.5612500309944153, "rewards/judge_quality/std": 0.1968638300895691, "rewards/total_composite/mean": 0.5114455819129944, "rewards/total_composite/std": 0.1362476795911789, "reward": 0.5114455819129944, "reward_std": 0.1362476795911789, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1806032508611679, "sampling/sampling_logp_difference/max": 2.0891857147216797, "sampling/importance_sampling_ratio/min": 0.2237635999917984, "sampling/importance_sampling_ratio/mean": 1.0257023572921753, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0947838053107262, "clip_ratio/low_mean": 0.09659948013722897, "clip_ratio/low_min": 0.09659948013722897, "clip_ratio/high_mean": 0.073857381939888, "clip_ratio/high_max": 0.073857381939888, "clip_ratio/region_mean": 0.17045686207711697, "reward_total_mean": 0.5114455819129944, "reward_meter_mean": 0.4592335820198059, "reward_meter_std": 0.3384995758533478, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9714906215667725, "reward_repeat_soft_std": 0.022578958421945572, "reward_judge_quality_mean": 0.5612500309944153, "reward_judge_quality_std": 0.1968638300895691, "reward_total_composite_mean": 0.5114455819129944, "reward_total_composite_std": 0.1362476795911789} {"timestamp_utc": "2026-04-13T09:10:06Z", "mode": "train", "global_step": 729, "epoch": 0.07322953289804118, "loss": 0.0938, "grad_norm": 13.141268730163574, "learning_rate": 7.793939393939394e-06, "num_tokens": 1288418.0, "completions/mean_length": 58.375, "completions/min_length": 46.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.375, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.6245149374008179, "rewards/meter/std": 0.3908325433731079, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8847557306289673, "rewards/repeat_soft/std": 0.12024150043725967, "rewards/judge_quality/mean": 0.53125, "rewards/judge_quality/std": 0.1970088928937912, "rewards/total_composite/mean": 0.4973753094673157, "rewards/total_composite/std": 0.25831887125968933, "reward": 0.4973753094673157, "reward_std": 0.2583189010620117, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1637883484363556, "sampling/sampling_logp_difference/max": 1.5596323013305664, "sampling/importance_sampling_ratio/min": 0.21021334826946259, "sampling/importance_sampling_ratio/mean": 1.0219950675964355, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8918347582221031, "clip_ratio/low_mean": 0.07465672492980957, "clip_ratio/low_min": 0.07465672492980957, "clip_ratio/high_mean": 0.07172060199081898, "clip_ratio/high_max": 0.07172060199081898, "clip_ratio/region_mean": 0.14637732692062855, "reward_total_mean": 0.4973753094673157, "reward_meter_mean": 0.6245149374008179, "reward_meter_std": 0.3908325433731079, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8847557306289673, "reward_repeat_soft_std": 0.12024150043725967, "reward_judge_quality_mean": 0.53125, "reward_judge_quality_std": 0.1970088928937912, "reward_total_composite_mean": 0.4973753094673157, "reward_total_composite_std": 0.25831887125968933} {"timestamp_utc": "2026-04-13T09:10:13Z", "mode": "train", "global_step": 730, "epoch": 0.07332998493219488, "loss": -0.039, "grad_norm": 14.46064567565918, "learning_rate": 7.790909090909092e-06, "num_tokens": 1290611.0, "completions/mean_length": 75.125, "completions/min_length": 59.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.125, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.568236231803894, "rewards/meter/std": 0.300004780292511, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9620645642280579, "rewards/repeat_soft/std": 0.029723232612013817, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.2030482292175293, "rewards/total_composite/mean": 0.5131289958953857, "rewards/total_composite/std": 0.10526996105909348, "reward": 0.5131289958953857, "reward_std": 0.10526996105909348, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18893152475357056, "sampling/sampling_logp_difference/max": 2.1716699600219727, "sampling/importance_sampling_ratio/min": 0.14820317924022675, "sampling/importance_sampling_ratio/mean": 1.0112147331237793, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1518063470721245, "clip_ratio/low_mean": 0.1104865912348032, "clip_ratio/low_min": 0.1104865912348032, "clip_ratio/high_mean": 0.04992319457232952, "clip_ratio/high_max": 0.04992319457232952, "clip_ratio/region_mean": 0.16040978580713272, "reward_total_mean": 0.5131289958953857, "reward_meter_mean": 0.568236231803894, "reward_meter_std": 0.300004780292511, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9620645642280579, "reward_repeat_soft_std": 0.029723232612013817, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.2030482292175293, "reward_total_composite_mean": 0.5131289958953857, "reward_total_composite_std": 0.10526996105909348} {"timestamp_utc": "2026-04-13T09:10:19Z", "mode": "train", "global_step": 731, "epoch": 0.07343043696634857, "loss": -0.0022, "grad_norm": 11.444204330444336, "learning_rate": 7.787878787878789e-06, "num_tokens": 1292653.0, "completions/mean_length": 86.25, "completions/min_length": 66.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.25, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.6808890104293823, "rewards/meter/std": 0.3985695242881775, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9651210308074951, "rewards/repeat_soft/std": 0.024766085669398308, "rewards/judge_quality/mean": 0.5612500309944153, "rewards/judge_quality/std": 0.1968638300895691, "rewards/total_composite/mean": 0.6168433427810669, "rewards/total_composite/std": 0.1998293399810791, "reward": 0.6168433427810669, "reward_std": 0.1998293399810791, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1904446929693222, "sampling/sampling_logp_difference/max": 2.048414707183838, "sampling/importance_sampling_ratio/min": 0.12893915176391602, "sampling/importance_sampling_ratio/mean": 1.0512608289718628, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6167262345552444, "clip_ratio/low_mean": 0.08131587691605091, "clip_ratio/low_min": 0.08131587691605091, "clip_ratio/high_mean": 0.1005734745413065, "clip_ratio/high_max": 0.1005734745413065, "clip_ratio/region_mean": 0.1818893514573574, "reward_total_mean": 0.6168433427810669, "reward_meter_mean": 0.6808890104293823, "reward_meter_std": 0.3985695242881775, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9651210308074951, "reward_repeat_soft_std": 0.024766085669398308, "reward_judge_quality_mean": 0.5612500309944153, "reward_judge_quality_std": 0.1968638300895691, "reward_total_composite_mean": 0.6168433427810669, "reward_total_composite_std": 0.1998293399810791} {"timestamp_utc": "2026-04-13T09:10:26Z", "mode": "train", "global_step": 732, "epoch": 0.07353088900050227, "loss": 0.036, "grad_norm": 9.532846450805664, "learning_rate": 7.784848484848484e-06, "num_tokens": 1294857.0, "completions/mean_length": 96.5, "completions/min_length": 88.0, "completions/max_length": 120.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.5, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 120.0, "rewards/meter/mean": 0.6556195020675659, "rewards/meter/std": 0.27973002195358276, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.88929283618927, "rewards/repeat_soft/std": 0.10339207947254181, "rewards/judge_quality/mean": 0.6699999570846558, "rewards/judge_quality/std": 0.22038927674293518, "rewards/total_composite/mean": 0.6277689933776855, "rewards/total_composite/std": 0.16709183156490326, "reward": 0.6277689933776855, "reward_std": 0.16709184646606445, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15185225009918213, "sampling/sampling_logp_difference/max": 1.9622669219970703, "sampling/importance_sampling_ratio/min": 0.14053946733474731, "sampling/importance_sampling_ratio/mean": 1.037898063659668, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2167585268616676, "clip_ratio/low_mean": 0.10452262032777071, "clip_ratio/low_min": 0.10452262032777071, "clip_ratio/high_mean": 0.04544956237077713, "clip_ratio/high_max": 0.04544956237077713, "clip_ratio/region_mean": 0.14997218269854784, "reward_total_mean": 0.6277689933776855, "reward_meter_mean": 0.6556195020675659, "reward_meter_std": 0.27973002195358276, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.88929283618927, "reward_repeat_soft_std": 0.10339207947254181, "reward_judge_quality_mean": 0.6699999570846558, "reward_judge_quality_std": 0.22038927674293518, "reward_total_composite_mean": 0.6277689933776855, "reward_total_composite_std": 0.16709183156490326} {"timestamp_utc": "2026-04-13T09:10:37Z", "mode": "train", "global_step": 733, "epoch": 0.07363134103465595, "loss": -0.0943, "grad_norm": 4.793450355529785, "learning_rate": 7.781818181818183e-06, "num_tokens": 1296295.0, "completions/mean_length": 101.75, "completions/min_length": 40.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 43.142860412597656, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.7315326929092407, "rewards/meter/std": 0.41042831540107727, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8619771003723145, "rewards/repeat_soft/std": 0.1800392121076584, "rewards/judge_quality/mean": 0.4737499952316284, "rewards/judge_quality/std": 0.254611074924469, "rewards/total_composite/mean": 0.516670823097229, "rewards/total_composite/std": 0.2958741784095764, "reward": 0.516670823097229, "reward_std": 0.29587414860725403, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13379375636577606, "sampling/sampling_logp_difference/max": 1.787715196609497, "sampling/importance_sampling_ratio/min": 0.16734206676483154, "sampling/importance_sampling_ratio/mean": 1.0093183517456055, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7671684697270393, "clip_ratio/low_mean": 0.023473837645724416, "clip_ratio/low_min": 0.023473837645724416, "clip_ratio/high_mean": 0.09144049324095249, "clip_ratio/high_max": 0.09144049324095249, "clip_ratio/region_mean": 0.11491433088667691, "reward_total_mean": 0.516670823097229, "reward_meter_mean": 0.7315326929092407, "reward_meter_std": 0.41042831540107727, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8619771003723145, "reward_repeat_soft_std": 0.1800392121076584, "reward_judge_quality_mean": 0.4737499952316284, "reward_judge_quality_std": 0.254611074924469, "reward_total_composite_mean": 0.516670823097229, "reward_total_composite_std": 0.2958741784095764} {"timestamp_utc": "2026-04-13T09:10:43Z", "mode": "train", "global_step": 734, "epoch": 0.07373179306880964, "loss": 0.0212, "grad_norm": 14.931814193725586, "learning_rate": 7.778787878787879e-06, "num_tokens": 1297938.0, "completions/mean_length": 39.375, "completions/min_length": 30.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.375, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.7002102136611938, "rewards/meter/std": 0.29298287630081177, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9946794509887695, "rewards/repeat_soft/std": 0.009923518635332584, "rewards/judge_quality/mean": 0.6150000095367432, "rewards/judge_quality/std": 0.25286927819252014, "rewards/total_composite/mean": 0.6478596925735474, "rewards/total_composite/std": 0.19373920559883118, "reward": 0.6478596925735474, "reward_std": 0.19373920559883118, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15438002347946167, "sampling/sampling_logp_difference/max": 1.7901697158813477, "sampling/importance_sampling_ratio/min": 0.16693183779716492, "sampling/importance_sampling_ratio/mean": 1.0184907913208008, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9812079221010208, "clip_ratio/low_mean": 0.08412531949579716, "clip_ratio/low_min": 0.08412531949579716, "clip_ratio/high_mean": 0.04582786979153752, "clip_ratio/high_max": 0.04582786979153752, "clip_ratio/region_mean": 0.12995318928733468, "reward_total_mean": 0.6478596925735474, "reward_meter_mean": 0.7002102136611938, "reward_meter_std": 0.29298287630081177, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9946794509887695, "reward_repeat_soft_std": 0.009923518635332584, "reward_judge_quality_mean": 0.6150000095367432, "reward_judge_quality_std": 0.25286927819252014, "reward_total_composite_mean": 0.6478596925735474, "reward_total_composite_std": 0.19373920559883118} {"timestamp_utc": "2026-04-13T09:10:49Z", "mode": "train", "global_step": 735, "epoch": 0.07383224510296334, "loss": -0.0292, "grad_norm": 17.531740188598633, "learning_rate": 7.775757575757576e-06, "num_tokens": 1299416.0, "completions/mean_length": 22.75, "completions/min_length": 18.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.75, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.6155325174331665, "rewards/meter/std": 0.34199750423431396, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9615821838378906, "rewards/repeat_soft/std": 0.002596014179289341, "rewards/judge_quality/mean": 0.48625001311302185, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.5228148698806763, "rewards/total_composite/std": 0.0878102108836174, "reward": 0.5228148698806763, "reward_std": 0.087810218334198, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19599971175193787, "sampling/sampling_logp_difference/max": 1.1137094497680664, "sampling/importance_sampling_ratio/min": 0.32833874225616455, "sampling/importance_sampling_ratio/mean": 1.085088849067688, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.71340611577034, "clip_ratio/low_mean": 0.08386752288788557, "clip_ratio/low_min": 0.08386752288788557, "clip_ratio/high_mean": 0.07213768130168319, "clip_ratio/high_max": 0.07213768130168319, "clip_ratio/region_mean": 0.15600520418956876, "reward_total_mean": 0.5228148698806763, "reward_meter_mean": 0.6155325174331665, "reward_meter_std": 0.34199750423431396, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9615821838378906, "reward_repeat_soft_std": 0.002596014179289341, "reward_judge_quality_mean": 0.48625001311302185, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.5228148698806763, "reward_total_composite_std": 0.0878102108836174} {"timestamp_utc": "2026-04-13T09:11:00Z", "mode": "train", "global_step": 736, "epoch": 0.07393269713711703, "loss": -0.1665, "grad_norm": 2.189173936843872, "learning_rate": 7.772727272727273e-06, "num_tokens": 1301239.0, "completions/mean_length": 191.875, "completions/min_length": 69.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 85.16667175292969, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.7461813688278198, "rewards/meter/std": 0.4131945073604584, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.8395068049430847, "rewards/repeat_soft/std": 0.11717653274536133, "rewards/judge_quality/mean": 0.5225000381469727, "rewards/judge_quality/std": 0.3446219563484192, "rewards/total_composite/mean": 0.5662407875061035, "rewards/total_composite/std": 0.3655031621456146, "reward": 0.5662407875061035, "reward_std": 0.36550313234329224, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1478935331106186, "sampling/sampling_logp_difference/max": 1.9637460708618164, "sampling/importance_sampling_ratio/min": 0.14033174514770508, "sampling/importance_sampling_ratio/mean": 1.0013198852539062, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7034620717167854, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.11650931276381016, "clip_ratio/high_max": 0.11650931276381016, "clip_ratio/region_mean": 0.11650931276381016, "reward_total_mean": 0.5662407875061035, "reward_meter_mean": 0.7461813688278198, "reward_meter_std": 0.4131945073604584, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.8395068049430847, "reward_repeat_soft_std": 0.11717653274536133, "reward_judge_quality_mean": 0.5225000381469727, "reward_judge_quality_std": 0.3446219563484192, "reward_total_composite_mean": 0.5662407875061035, "reward_total_composite_std": 0.3655031621456146} {"timestamp_utc": "2026-04-13T09:11:06Z", "mode": "train", "global_step": 737, "epoch": 0.07403314917127071, "loss": 0.1024, "grad_norm": 21.48760414123535, "learning_rate": 7.76969696969697e-06, "num_tokens": 1302578.0, "completions/mean_length": 25.375, "completions/min_length": 21.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.375, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.31592369079589844, "rewards/meter/std": 0.3158826231956482, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9594065546989441, "rewards/repeat_soft/std": 0.008749538101255894, "rewards/judge_quality/mean": 0.5049999952316284, "rewards/judge_quality/std": 0.16801361739635468, "rewards/total_composite/mean": 0.45355793833732605, "rewards/total_composite/std": 0.1119605228304863, "reward": 0.45355793833732605, "reward_std": 0.1119605302810669, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21762022376060486, "sampling/sampling_logp_difference/max": 1.8392524719238281, "sampling/importance_sampling_ratio/min": 0.15893620252609253, "sampling/importance_sampling_ratio/mean": 1.0408340692520142, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0567884147167206, "clip_ratio/low_mean": 0.08554933313280344, "clip_ratio/low_min": 0.08554933313280344, "clip_ratio/high_mean": 0.07183441706001759, "clip_ratio/high_max": 0.07183441706001759, "clip_ratio/region_mean": 0.15738375019282103, "reward_total_mean": 0.45355793833732605, "reward_meter_mean": 0.31592369079589844, "reward_meter_std": 0.3158826231956482, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9594065546989441, "reward_repeat_soft_std": 0.008749538101255894, "reward_judge_quality_mean": 0.5049999952316284, "reward_judge_quality_std": 0.16801361739635468, "reward_total_composite_mean": 0.45355793833732605, "reward_total_composite_std": 0.1119605228304863} {"timestamp_utc": "2026-04-13T09:11:12Z", "mode": "train", "global_step": 738, "epoch": 0.07413360120542441, "loss": 0.0724, "grad_norm": 13.87582015991211, "learning_rate": 7.766666666666666e-06, "num_tokens": 1304338.0, "completions/mean_length": 42.0, "completions/min_length": 29.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.0, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.6752654314041138, "rewards/meter/std": 0.32563450932502747, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9729601144790649, "rewards/repeat_soft/std": 0.03770548850297928, "rewards/judge_quality/mean": 0.8612500429153442, "rewards/judge_quality/std": 0.16617010533809662, "rewards/total_composite/mean": 0.7150299549102783, "rewards/total_composite/std": 0.18782474100589752, "reward": 0.7150299549102783, "reward_std": 0.18782474100589752, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16547410190105438, "sampling/sampling_logp_difference/max": 1.6455574035644531, "sampling/importance_sampling_ratio/min": 0.1929050087928772, "sampling/importance_sampling_ratio/mean": 0.9987207055091858, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8475840240716934, "clip_ratio/low_mean": 0.0568147050216794, "clip_ratio/low_min": 0.0568147050216794, "clip_ratio/high_mean": 0.08551109861582518, "clip_ratio/high_max": 0.08551109861582518, "clip_ratio/region_mean": 0.14232580363750458, "reward_total_mean": 0.7150299549102783, "reward_meter_mean": 0.6752654314041138, "reward_meter_std": 0.32563450932502747, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9729601144790649, "reward_repeat_soft_std": 0.03770548850297928, "reward_judge_quality_mean": 0.8612500429153442, "reward_judge_quality_std": 0.16617010533809662, "reward_total_composite_mean": 0.7150299549102783, "reward_total_composite_std": 0.18782474100589752} {"timestamp_utc": "2026-04-13T09:11:18Z", "mode": "train", "global_step": 739, "epoch": 0.0742340532395781, "loss": 0.0354, "grad_norm": 11.87075138092041, "learning_rate": 7.763636363636364e-06, "num_tokens": 1306516.0, "completions/mean_length": 95.25, "completions/min_length": 72.0, "completions/max_length": 112.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.25, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 112.0, "rewards/meter/mean": 0.6119871139526367, "rewards/meter/std": 0.3081716001033783, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9510897397994995, "rewards/repeat_soft/std": 0.040236297994852066, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.20469054579734802, "rewards/total_composite/mean": 0.5248219966888428, "rewards/total_composite/std": 0.1047632247209549, "reward": 0.5248219966888428, "reward_std": 0.10476323962211609, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16723228991031647, "sampling/sampling_logp_difference/max": 1.7635916471481323, "sampling/importance_sampling_ratio/min": 0.17142803966999054, "sampling/importance_sampling_ratio/mean": 1.0197486877441406, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0619538724422455, "clip_ratio/low_mean": 0.11025430075824261, "clip_ratio/low_min": 0.11025430075824261, "clip_ratio/high_mean": 0.06122303195297718, "clip_ratio/high_max": 0.06122303195297718, "clip_ratio/region_mean": 0.1714773327112198, "reward_total_mean": 0.5248219966888428, "reward_meter_mean": 0.6119871139526367, "reward_meter_std": 0.3081716001033783, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9510897397994995, "reward_repeat_soft_std": 0.040236297994852066, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.20469054579734802, "reward_total_composite_mean": 0.5248219966888428, "reward_total_composite_std": 0.1047632247209549} {"timestamp_utc": "2026-04-13T09:11:30Z", "mode": "train", "global_step": 740, "epoch": 0.0743345052737318, "loss": -0.0624, "grad_norm": 4.005544185638428, "learning_rate": 7.76060606060606e-06, "num_tokens": 1307801.0, "completions/mean_length": 80.625, "completions/min_length": 16.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 19.0, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 23.0, "rewards/meter/mean": 0.39341825246810913, "rewards/meter/std": 0.41285014152526855, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.627500057220459, "rewards/judge_quality/std": 0.3366537392139435, "rewards/total_composite/mean": 0.4925876557826996, "rewards/total_composite/std": 0.3017560839653015, "reward": 0.4925876557826996, "reward_std": 0.3017560541629791, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17517143487930298, "sampling/sampling_logp_difference/max": 1.2181692123413086, "sampling/importance_sampling_ratio/min": 0.29577115178108215, "sampling/importance_sampling_ratio/mean": 1.0153645277023315, "sampling/importance_sampling_ratio/max": 1.9104666709899902, "entropy": 1.3799055814743042, "clip_ratio/low_mean": 0.07118055690079927, "clip_ratio/low_min": 0.07118055690079927, "clip_ratio/high_mean": 0.06757382769137621, "clip_ratio/high_max": 0.06757382769137621, "clip_ratio/region_mean": 0.13875438459217548, "reward_total_mean": 0.4925876557826996, "reward_meter_mean": 0.39341825246810913, "reward_meter_std": 0.41285014152526855, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.627500057220459, "reward_judge_quality_std": 0.3366537392139435, "reward_total_composite_mean": 0.4925876557826996, "reward_total_composite_std": 0.3017560839653015} {"timestamp_utc": "2026-04-13T09:11:41Z", "mode": "train", "global_step": 741, "epoch": 0.07443495730788549, "loss": -0.1253, "grad_norm": 3.169534683227539, "learning_rate": 7.757575757575758e-06, "num_tokens": 1309498.0, "completions/mean_length": 107.125, "completions/min_length": 34.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 49.28571701049805, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9664974808692932, "rewards/meter/std": 0.01657053641974926, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9849109053611755, "rewards/repeat_soft/std": 0.03397379815578461, "rewards/judge_quality/mean": 0.48124998807907104, "rewards/judge_quality/std": 0.2582046389579773, "rewards/total_composite/mean": 0.6017608642578125, "rewards/total_composite/std": 0.27139541506767273, "reward": 0.6017608642578125, "reward_std": 0.27139541506767273, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1672983467578888, "sampling/sampling_logp_difference/max": 1.6825370788574219, "sampling/importance_sampling_ratio/min": 0.1859017312526703, "sampling/importance_sampling_ratio/mean": 1.0501664876937866, "sampling/importance_sampling_ratio/max": 1.9643293619155884, "entropy": 1.2438460141420364, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.159818759188056, "clip_ratio/high_max": 0.159818759188056, "clip_ratio/region_mean": 0.159818759188056, "reward_total_mean": 0.6017608642578125, "reward_meter_mean": 0.9664974808692932, "reward_meter_std": 0.01657053641974926, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9849109053611755, "reward_repeat_soft_std": 0.03397379815578461, "reward_judge_quality_mean": 0.48124998807907104, "reward_judge_quality_std": 0.2582046389579773, "reward_total_composite_mean": 0.6017608642578125, "reward_total_composite_std": 0.27139541506767273} {"timestamp_utc": "2026-04-13T09:11:52Z", "mode": "train", "global_step": 742, "epoch": 0.07453540934203917, "loss": -0.0849, "grad_norm": 3.7173469066619873, "learning_rate": 7.754545454545455e-06, "num_tokens": 1311239.0, "completions/mean_length": 105.625, "completions/min_length": 41.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 47.57143020629883, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.46908485889434814, "rewards/meter/std": 0.44270479679107666, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9901416301727295, "rewards/repeat_soft/std": 0.013197098858654499, "rewards/judge_quality/mean": 0.5137500166893005, "rewards/judge_quality/std": 0.2563444972038269, "rewards/total_composite/mean": 0.4840984344482422, "rewards/total_composite/std": 0.26524755358695984, "reward": 0.4840984344482422, "reward_std": 0.26524755358695984, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20756895840168, "sampling/sampling_logp_difference/max": 1.9614200592041016, "sampling/importance_sampling_ratio/min": 0.14065852761268616, "sampling/importance_sampling_ratio/mean": 1.032393217086792, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4433472752571106, "clip_ratio/low_mean": 0.06576655060052872, "clip_ratio/low_min": 0.06576655060052872, "clip_ratio/high_mean": 0.09691833518445492, "clip_ratio/high_max": 0.09691833518445492, "clip_ratio/region_mean": 0.16268488578498363, "reward_total_mean": 0.4840984344482422, "reward_meter_mean": 0.46908485889434814, "reward_meter_std": 0.44270479679107666, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9901416301727295, "reward_repeat_soft_std": 0.013197098858654499, "reward_judge_quality_mean": 0.5137500166893005, "reward_judge_quality_std": 0.2563444972038269, "reward_total_composite_mean": 0.4840984344482422, "reward_total_composite_std": 0.26524755358695984} {"timestamp_utc": "2026-04-13T09:11:58Z", "mode": "train", "global_step": 743, "epoch": 0.07463586137619287, "loss": -0.0189, "grad_norm": 12.860284805297852, "learning_rate": 7.751515151515153e-06, "num_tokens": 1313094.0, "completions/mean_length": 58.875, "completions/min_length": 40.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.875, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.8150298595428467, "rewards/meter/std": 0.29089590907096863, "rewards/count_adherence/mean": 0.84375, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9740594625473022, "rewards/repeat_soft/std": 0.016727657988667488, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.5241194367408752, "rewards/total_composite/std": 0.0947740450501442, "reward": 0.5241194367408752, "reward_std": 0.09477405250072479, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20211203396320343, "sampling/sampling_logp_difference/max": 3.094081401824951, "sampling/importance_sampling_ratio/min": 0.144920215010643, "sampling/importance_sampling_ratio/mean": 1.024543046951294, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.351658508181572, "clip_ratio/low_mean": 0.07901785522699356, "clip_ratio/low_min": 0.07901785522699356, "clip_ratio/high_mean": 0.11662971321493387, "clip_ratio/high_max": 0.11662971321493387, "clip_ratio/region_mean": 0.19564756844192743, "reward_total_mean": 0.5241194367408752, "reward_meter_mean": 0.8150298595428467, "reward_meter_std": 0.29089590907096863, "reward_count_adherence_mean": 0.84375, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9740594625473022, "reward_repeat_soft_std": 0.016727657988667488, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.5241194367408752, "reward_total_composite_std": 0.0947740450501442} {"timestamp_utc": "2026-04-13T09:12:05Z", "mode": "train", "global_step": 744, "epoch": 0.07473631341034656, "loss": 0.0216, "grad_norm": 9.15574836730957, "learning_rate": 7.74848484848485e-06, "num_tokens": 1315335.0, "completions/mean_length": 113.125, "completions/min_length": 94.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 113.125, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.7801750898361206, "rewards/meter/std": 0.27828752994537354, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9237654805183411, "rewards/repeat_soft/std": 0.04357816278934479, "rewards/judge_quality/mean": 0.6449999809265137, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.5798206329345703, "rewards/total_composite/std": 0.2641700506210327, "reward": 0.5798206329345703, "reward_std": 0.2641700506210327, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1706664115190506, "sampling/sampling_logp_difference/max": 2.086062431335449, "sampling/importance_sampling_ratio/min": 0.12417511641979218, "sampling/importance_sampling_ratio/mean": 1.0122138261795044, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2584090828895569, "clip_ratio/low_mean": 0.0580312255769968, "clip_ratio/low_min": 0.0580312255769968, "clip_ratio/high_mean": 0.1189239863306284, "clip_ratio/high_max": 0.1189239863306284, "clip_ratio/region_mean": 0.1769552119076252, "reward_total_mean": 0.5798206329345703, "reward_meter_mean": 0.7801750898361206, "reward_meter_std": 0.27828752994537354, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9237654805183411, "reward_repeat_soft_std": 0.04357816278934479, "reward_judge_quality_mean": 0.6449999809265137, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.5798206329345703, "reward_total_composite_std": 0.2641700506210327} {"timestamp_utc": "2026-04-13T09:12:11Z", "mode": "train", "global_step": 745, "epoch": 0.07483676544450026, "loss": 0.0347, "grad_norm": 17.920703887939453, "learning_rate": 7.745454545454545e-06, "num_tokens": 1316895.0, "completions/mean_length": 43.0, "completions/min_length": 38.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.0, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9510095119476318, "rewards/meter/std": 0.10056521743535995, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.968070387840271, "rewards/repeat_soft/std": 0.03921138867735863, "rewards/judge_quality/mean": 0.7987500429153442, "rewards/judge_quality/std": 0.22465452551841736, "rewards/total_composite/mean": 0.8477826714515686, "rewards/total_composite/std": 0.16634705662727356, "reward": 0.8477826714515686, "reward_std": 0.16634705662727356, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18730603158473969, "sampling/sampling_logp_difference/max": 1.5259227752685547, "sampling/importance_sampling_ratio/min": 0.21742035448551178, "sampling/importance_sampling_ratio/mean": 1.0188100337982178, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.546481341123581, "clip_ratio/low_mean": 0.029415836557745934, "clip_ratio/low_min": 0.029415836557745934, "clip_ratio/high_mean": 0.14870378375053406, "clip_ratio/high_max": 0.14870378375053406, "clip_ratio/region_mean": 0.17811962030828, "reward_total_mean": 0.8477826714515686, "reward_meter_mean": 0.9510095119476318, "reward_meter_std": 0.10056521743535995, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.968070387840271, "reward_repeat_soft_std": 0.03921138867735863, "reward_judge_quality_mean": 0.7987500429153442, "reward_judge_quality_std": 0.22465452551841736, "reward_total_composite_mean": 0.8477826714515686, "reward_total_composite_std": 0.16634705662727356} {"timestamp_utc": "2026-04-13T09:12:17Z", "mode": "train", "global_step": 746, "epoch": 0.07493721747865394, "loss": -0.0258, "grad_norm": 16.017648696899414, "learning_rate": 7.742424242424244e-06, "num_tokens": 1318476.0, "completions/mean_length": 43.625, "completions/min_length": 38.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.625, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.7205237746238708, "rewards/meter/std": 0.39658796787261963, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9746197462081909, "rewards/repeat_soft/std": 0.03359459713101387, "rewards/judge_quality/mean": 0.4937500059604645, "rewards/judge_quality/std": 0.1728696972131729, "rewards/total_composite/mean": 0.5865886211395264, "rewards/total_composite/std": 0.17276552319526672, "reward": 0.5865886211395264, "reward_std": 0.17276552319526672, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1927284151315689, "sampling/sampling_logp_difference/max": 1.848165512084961, "sampling/importance_sampling_ratio/min": 0.15752588212490082, "sampling/importance_sampling_ratio/mean": 0.9987056851387024, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4613386616110802, "clip_ratio/low_mean": 0.04135736916214228, "clip_ratio/low_min": 0.04135736916214228, "clip_ratio/high_mean": 0.09174805972725153, "clip_ratio/high_max": 0.09174805972725153, "clip_ratio/region_mean": 0.1331054288893938, "reward_total_mean": 0.5865886211395264, "reward_meter_mean": 0.7205237746238708, "reward_meter_std": 0.39658796787261963, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9746197462081909, "reward_repeat_soft_std": 0.03359459713101387, "reward_judge_quality_mean": 0.4937500059604645, "reward_judge_quality_std": 0.1728696972131729, "reward_total_composite_mean": 0.5865886211395264, "reward_total_composite_std": 0.17276552319526672} {"timestamp_utc": "2026-04-13T09:12:23Z", "mode": "train", "global_step": 747, "epoch": 0.07503766951280763, "loss": 0.0372, "grad_norm": 21.14970588684082, "learning_rate": 7.73939393939394e-06, "num_tokens": 1319893.0, "completions/mean_length": 23.125, "completions/min_length": 17.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.125, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.6219501495361328, "rewards/meter/std": 0.5007457733154297, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8457368612289429, "rewards/repeat_soft/std": 0.09808874130249023, "rewards/judge_quality/mean": 0.5562499761581421, "rewards/judge_quality/std": 0.31717222929000854, "rewards/total_composite/mean": 0.5735586881637573, "rewards/total_composite/std": 0.3366280794143677, "reward": 0.5735586881637573, "reward_std": 0.3366280794143677, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19304467737674713, "sampling/sampling_logp_difference/max": 1.6328563690185547, "sampling/importance_sampling_ratio/min": 0.19537071883678436, "sampling/importance_sampling_ratio/mean": 1.0243732929229736, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4444321170449257, "clip_ratio/low_mean": 0.07059548143297434, "clip_ratio/low_min": 0.07059548143297434, "clip_ratio/high_mean": 0.10715579986572266, "clip_ratio/high_max": 0.10715579986572266, "clip_ratio/region_mean": 0.177751281298697, "reward_total_mean": 0.5735586881637573, "reward_meter_mean": 0.6219501495361328, "reward_meter_std": 0.5007457733154297, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8457368612289429, "reward_repeat_soft_std": 0.09808874130249023, "reward_judge_quality_mean": 0.5562499761581421, "reward_judge_quality_std": 0.31717222929000854, "reward_total_composite_mean": 0.5735586881637573, "reward_total_composite_std": 0.3366280794143677} {"timestamp_utc": "2026-04-13T09:12:29Z", "mode": "train", "global_step": 748, "epoch": 0.07513812154696133, "loss": 0.0543, "grad_norm": 20.271127700805664, "learning_rate": 7.736363636363637e-06, "num_tokens": 1321439.0, "completions/mean_length": 37.25, "completions/min_length": 32.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.25, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.9286379814147949, "rewards/meter/std": 0.13010568916797638, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9717549085617065, "rewards/repeat_soft/std": 0.019136307761073112, "rewards/judge_quality/mean": 0.6112500429153442, "rewards/judge_quality/std": 0.25587037205696106, "rewards/total_composite/mean": 0.655158519744873, "rewards/total_composite/std": 0.3113307058811188, "reward": 0.655158519744873, "reward_std": 0.3113307058811188, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17222309112548828, "sampling/sampling_logp_difference/max": 2.0556488037109375, "sampling/importance_sampling_ratio/min": 0.12800975143909454, "sampling/importance_sampling_ratio/mean": 1.0102615356445312, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3393578231334686, "clip_ratio/low_mean": 0.09474449045956135, "clip_ratio/low_min": 0.09474449045956135, "clip_ratio/high_mean": 0.07014034502208233, "clip_ratio/high_max": 0.07014034502208233, "clip_ratio/region_mean": 0.16488483548164368, "reward_total_mean": 0.655158519744873, "reward_meter_mean": 0.9286379814147949, "reward_meter_std": 0.13010568916797638, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9717549085617065, "reward_repeat_soft_std": 0.019136307761073112, "reward_judge_quality_mean": 0.6112500429153442, "reward_judge_quality_std": 0.25587037205696106, "reward_total_composite_mean": 0.655158519744873, "reward_total_composite_std": 0.3113307058811188} {"timestamp_utc": "2026-04-13T09:12:41Z", "mode": "train", "global_step": 749, "epoch": 0.07523857358111502, "loss": -0.1137, "grad_norm": 4.881791114807129, "learning_rate": 7.733333333333334e-06, "num_tokens": 1323294.0, "completions/mean_length": 120.875, "completions/min_length": 52.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 65.0, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.41662102937698364, "rewards/meter/std": 0.33569326996803284, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9671957492828369, "rewards/repeat_soft/std": 0.020684966817498207, "rewards/judge_quality/mean": 0.5900000333786011, "rewards/judge_quality/std": 0.317265123128891, "rewards/total_composite/mean": 0.4998413622379303, "rewards/total_composite/std": 0.27943938970565796, "reward": 0.4998413622379303, "reward_std": 0.27943938970565796, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20088011026382446, "sampling/sampling_logp_difference/max": 3.438444137573242, "sampling/importance_sampling_ratio/min": 0.03211461380124092, "sampling/importance_sampling_ratio/mean": 0.993660569190979, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.191350631415844, "clip_ratio/low_mean": 0.06448888964951038, "clip_ratio/low_min": 0.06448888964951038, "clip_ratio/high_mean": 0.08905814588069916, "clip_ratio/high_max": 0.08905814588069916, "clip_ratio/region_mean": 0.15354703553020954, "reward_total_mean": 0.4998413622379303, "reward_meter_mean": 0.41662102937698364, "reward_meter_std": 0.33569326996803284, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9671957492828369, "reward_repeat_soft_std": 0.020684966817498207, "reward_judge_quality_mean": 0.5900000333786011, "reward_judge_quality_std": 0.317265123128891, "reward_total_composite_mean": 0.4998413622379303, "reward_total_composite_std": 0.27943938970565796} {"timestamp_utc": "2026-04-13T09:12:52Z", "mode": "train", "global_step": 750, "epoch": 0.07533902561526871, "loss": -0.0773, "grad_norm": 4.638439178466797, "learning_rate": 7.730303030303032e-06, "num_tokens": 1325106.0, "completions/mean_length": 95.5, "completions/min_length": 34.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 36.0, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.6533169746398926, "rewards/meter/std": 0.4125302731990814, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.992682933807373, "rewards/repeat_soft/std": 0.010398227721452713, "rewards/judge_quality/mean": 0.6312500238418579, "rewards/judge_quality/std": 0.33417007327079773, "rewards/total_composite/mean": 0.622667133808136, "rewards/total_composite/std": 0.32134515047073364, "reward": 0.622667133808136, "reward_std": 0.32134518027305603, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18471704423427582, "sampling/sampling_logp_difference/max": 1.7912163734436035, "sampling/importance_sampling_ratio/min": 0.16675721108913422, "sampling/importance_sampling_ratio/mean": 1.0475273132324219, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9543810561299324, "clip_ratio/low_mean": 0.05844298284500837, "clip_ratio/low_min": 0.05844298284500837, "clip_ratio/high_mean": 0.07930672587826848, "clip_ratio/high_max": 0.07930672587826848, "clip_ratio/region_mean": 0.13774970872327685, "reward_total_mean": 0.622667133808136, "reward_meter_mean": 0.6533169746398926, "reward_meter_std": 0.4125302731990814, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.992682933807373, "reward_repeat_soft_std": 0.010398227721452713, "reward_judge_quality_mean": 0.6312500238418579, "reward_judge_quality_std": 0.33417007327079773, "reward_total_composite_mean": 0.622667133808136, "reward_total_composite_std": 0.32134515047073364} {"timestamp_utc": "2026-04-13T09:13:52Z", "mode": "eval", "global_step": 750, "epoch": 0.07533902561526871, "eval_loss": NaN, "eval_runtime": 59.8464, "eval_samples_per_second": 1.337, "eval_steps_per_second": 0.167, "eval_num_tokens": 1325106.0, "eval_completions/mean_length": 100.5, "eval_completions/min_length": 37.2, "eval_completions/max_length": 281.9, "eval_completions/clipped_ratio": 0.0625, "eval_completions/mean_terminated_length": 72.86488265991211, "eval_completions/min_terminated_length": 37.2, "eval_completions/max_terminated_length": 125.0, "eval_rewards/meter/mean": 0.6636147797107697, "eval_rewards/meter/std": 0.37332658767700194, "eval_rewards/count_adherence/mean": 0.9685416698455811, "eval_rewards/count_adherence/std": 0.08160217814147472, "eval_rewards/hard_gate/mean": 0.9125, "eval_rewards/hard_gate/std": 0.2230676978826523, "eval_rewards/repeat_soft/mean": 0.9539894819259643, "eval_rewards/repeat_soft/std": 0.05015313681215048, "eval_rewards/judge_quality/mean": 0.5278750061988831, "eval_rewards/judge_quality/std": 0.22390185445547103, "eval_rewards/total_composite/mean": 0.5406864643096924, "eval_rewards/total_composite/std": 0.22603418827056884, "eval_reward": 0.5406864643096924, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.09669260457158088, "eval_sampling/sampling_logp_difference/max": 1.0985046863555907, "eval_sampling/importance_sampling_ratio/min": 0.34071766436100004, "eval_sampling/importance_sampling_ratio/mean": 1.030054771900177, "eval_sampling/importance_sampling_ratio/max": 1.5093917369842529, "eval_entropy": 1.175079160928726, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5406864643096924, "eval_reward_meter_mean": 0.6636147797107697, "eval_reward_meter_std": 0.37332658767700194, "eval_reward_count_adherence_mean": 0.9685416698455811, "eval_reward_count_adherence_std": 0.08160217814147472, "eval_reward_hard_gate_mean": 0.9125, "eval_reward_hard_gate_std": 0.2230676978826523, "eval_reward_repeat_soft_mean": 0.9539894819259643, "eval_reward_repeat_soft_std": 0.05015313681215048, "eval_reward_judge_quality_mean": 0.5278750061988831, "eval_reward_judge_quality_std": 0.22390185445547103, "eval_reward_total_composite_mean": 0.5406864643096924, "eval_reward_total_composite_std": 0.22603418827056884} {"timestamp_utc": "2026-04-13T09:14:06Z", "mode": "train", "global_step": 751, "epoch": 0.0754394776494224, "loss": -0.1672, "grad_norm": 3.318563938140869, "learning_rate": 7.727272727272727e-06, "num_tokens": 1326802.0, "completions/mean_length": 121.0, "completions/min_length": 38.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 65.14286041259766, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.8778524398803711, "rewards/meter/std": 0.2851432263851166, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9820973873138428, "rewards/repeat_soft/std": 0.038947753608226776, "rewards/judge_quality/mean": 0.5774999856948853, "rewards/judge_quality/std": 0.25616681575775146, "rewards/total_composite/mean": 0.6581735014915466, "rewards/total_composite/std": 0.2878771722316742, "reward": 0.6581735014915466, "reward_std": 0.2878771722316742, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1561141312122345, "sampling/sampling_logp_difference/max": 1.4459128379821777, "sampling/importance_sampling_ratio/min": 0.23553098738193512, "sampling/importance_sampling_ratio/mean": 1.0225884914398193, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7705533057451248, "clip_ratio/low_mean": 0.032418000511825085, "clip_ratio/low_min": 0.032418000511825085, "clip_ratio/high_mean": 0.07891025021672249, "clip_ratio/high_max": 0.07891025021672249, "clip_ratio/region_mean": 0.11132825072854757, "reward_total_mean": 0.6581735014915466, "reward_meter_mean": 0.8778524398803711, "reward_meter_std": 0.2851432263851166, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9820973873138428, "reward_repeat_soft_std": 0.038947753608226776, "reward_judge_quality_mean": 0.5774999856948853, "reward_judge_quality_std": 0.25616681575775146, "reward_total_composite_mean": 0.6581735014915466, "reward_total_composite_std": 0.2878771722316742} {"timestamp_utc": "2026-04-13T09:14:17Z", "mode": "train", "global_step": 752, "epoch": 0.07553992968357609, "loss": -0.1935, "grad_norm": 2.901785373687744, "learning_rate": 7.724242424242424e-06, "num_tokens": 1329034.0, "completions/mean_length": 210.0, "completions/min_length": 93.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 109.33333587646484, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 137.0, "rewards/meter/mean": 0.6406359076499939, "rewards/meter/std": 0.41214004158973694, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9574401378631592, "rewards/repeat_soft/std": 0.0687209814786911, "rewards/judge_quality/mean": 0.5112500190734863, "rewards/judge_quality/std": 0.3117433786392212, "rewards/total_composite/mean": 0.5163625478744507, "rewards/total_composite/std": 0.3412361741065979, "reward": 0.5163625478744507, "reward_std": 0.3412361741065979, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1524454653263092, "sampling/sampling_logp_difference/max": 1.9500782489776611, "sampling/importance_sampling_ratio/min": 0.14226293563842773, "sampling/importance_sampling_ratio/mean": 1.0099399089813232, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0000339895486832, "clip_ratio/low_mean": 0.012376237660646439, "clip_ratio/low_min": 0.012376237660646439, "clip_ratio/high_mean": 0.10809026099741459, "clip_ratio/high_max": 0.10809026099741459, "clip_ratio/region_mean": 0.12046649865806103, "reward_total_mean": 0.5163625478744507, "reward_meter_mean": 0.6406359076499939, "reward_meter_std": 0.41214004158973694, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9574401378631592, "reward_repeat_soft_std": 0.0687209814786911, "reward_judge_quality_mean": 0.5112500190734863, "reward_judge_quality_std": 0.3117433786392212, "reward_total_composite_mean": 0.5163625478744507, "reward_total_composite_std": 0.3412361741065979} {"timestamp_utc": "2026-04-13T09:14:24Z", "mode": "train", "global_step": 753, "epoch": 0.07564038171772978, "loss": 0.0651, "grad_norm": 13.571142196655273, "learning_rate": 7.721212121212122e-06, "num_tokens": 1330752.0, "completions/mean_length": 44.75, "completions/min_length": 38.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.75, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.7906677722930908, "rewards/meter/std": 0.3044937551021576, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9969684481620789, "rewards/repeat_soft/std": 0.00582959083840251, "rewards/judge_quality/mean": 0.6100000143051147, "rewards/judge_quality/std": 0.24628673493862152, "rewards/total_composite/mean": 0.6356469392776489, "rewards/total_composite/std": 0.1697525978088379, "reward": 0.6356469392776489, "reward_std": 0.1697525978088379, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1771460473537445, "sampling/sampling_logp_difference/max": 1.5940003395080566, "sampling/importance_sampling_ratio/min": 0.20311148464679718, "sampling/importance_sampling_ratio/mean": 1.0426604747772217, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4844245836138725, "clip_ratio/low_mean": 0.10905131604522467, "clip_ratio/low_min": 0.10905131604522467, "clip_ratio/high_mean": 0.06427770107984543, "clip_ratio/high_max": 0.06427770107984543, "clip_ratio/region_mean": 0.1733290171250701, "reward_total_mean": 0.6356469392776489, "reward_meter_mean": 0.7906677722930908, "reward_meter_std": 0.3044937551021576, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9969684481620789, "reward_repeat_soft_std": 0.00582959083840251, "reward_judge_quality_mean": 0.6100000143051147, "reward_judge_quality_std": 0.24628673493862152, "reward_total_composite_mean": 0.6356469392776489, "reward_total_composite_std": 0.1697525978088379} {"timestamp_utc": "2026-04-13T09:14:30Z", "mode": "train", "global_step": 754, "epoch": 0.07574083375188348, "loss": 0.0222, "grad_norm": 11.563011169433594, "learning_rate": 7.718181818181819e-06, "num_tokens": 1332278.0, "completions/mean_length": 45.75, "completions/min_length": 40.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.75, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.31257134675979614, "rewards/meter/std": 0.32782265543937683, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9939672946929932, "rewards/repeat_soft/std": 0.006959215737879276, "rewards/judge_quality/mean": 0.5350000262260437, "rewards/judge_quality/std": 0.18431341648101807, "rewards/total_composite/mean": 0.40538159012794495, "rewards/total_composite/std": 0.18614186346530914, "reward": 0.40538159012794495, "reward_std": 0.18614186346530914, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14600770175457, "sampling/sampling_logp_difference/max": 1.466710090637207, "sampling/importance_sampling_ratio/min": 0.23068316280841827, "sampling/importance_sampling_ratio/mean": 1.0287079811096191, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9013760164380074, "clip_ratio/low_mean": 0.04290668945759535, "clip_ratio/low_min": 0.04290668945759535, "clip_ratio/high_mean": 0.07478864770382643, "clip_ratio/high_max": 0.07478864770382643, "clip_ratio/region_mean": 0.11769533716142178, "reward_total_mean": 0.40538159012794495, "reward_meter_mean": 0.31257134675979614, "reward_meter_std": 0.32782265543937683, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9939672946929932, "reward_repeat_soft_std": 0.006959215737879276, "reward_judge_quality_mean": 0.5350000262260437, "reward_judge_quality_std": 0.18431341648101807, "reward_total_composite_mean": 0.40538159012794495, "reward_total_composite_std": 0.18614186346530914} {"timestamp_utc": "2026-04-13T09:14:41Z", "mode": "train", "global_step": 755, "epoch": 0.07584128578603717, "loss": -0.0203, "grad_norm": 6.581203937530518, "learning_rate": 7.715151515151516e-06, "num_tokens": 1333897.0, "completions/mean_length": 90.375, "completions/min_length": 24.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 30.142858505249023, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.3430681824684143, "rewards/meter/std": 0.3828832507133484, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9945133328437805, "rewards/repeat_soft/std": 0.004590123891830444, "rewards/judge_quality/mean": 0.5400000214576721, "rewards/judge_quality/std": 0.3381884694099426, "rewards/total_composite/mean": 0.3937174677848816, "rewards/total_composite/std": 0.28141385316848755, "reward": 0.3937174677848816, "reward_std": 0.28141382336616516, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17463988065719604, "sampling/sampling_logp_difference/max": 1.883415937423706, "sampling/importance_sampling_ratio/min": 0.15206974744796753, "sampling/importance_sampling_ratio/mean": 0.9966902732849121, "sampling/importance_sampling_ratio/max": 1.9669967889785767, "entropy": 0.890459232032299, "clip_ratio/low_mean": 0.044477523770183325, "clip_ratio/low_min": 0.044477523770183325, "clip_ratio/high_mean": 0.07569331582635641, "clip_ratio/high_max": 0.07569331582635641, "clip_ratio/region_mean": 0.12017083959653974, "reward_total_mean": 0.3937174677848816, "reward_meter_mean": 0.3430681824684143, "reward_meter_std": 0.3828832507133484, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9945133328437805, "reward_repeat_soft_std": 0.004590123891830444, "reward_judge_quality_mean": 0.5400000214576721, "reward_judge_quality_std": 0.3381884694099426, "reward_total_composite_mean": 0.3937174677848816, "reward_total_composite_std": 0.28141385316848755} {"timestamp_utc": "2026-04-13T09:14:47Z", "mode": "train", "global_step": 756, "epoch": 0.07594173782019085, "loss": 0.0408, "grad_norm": 14.017119407653809, "learning_rate": 7.712121212121213e-06, "num_tokens": 1335650.0, "completions/mean_length": 52.125, "completions/min_length": 47.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.125, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.4777580499649048, "rewards/meter/std": 0.36191046237945557, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9877919554710388, "rewards/repeat_soft/std": 0.027078699320554733, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.22720351815223694, "rewards/total_composite/mean": 0.5244680047035217, "rewards/total_composite/std": 0.16301079094409943, "reward": 0.5244680047035217, "reward_std": 0.16301079094409943, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.175713449716568, "sampling/sampling_logp_difference/max": 2.6853513717651367, "sampling/importance_sampling_ratio/min": 0.06819722801446915, "sampling/importance_sampling_ratio/mean": 1.02373206615448, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.249506413936615, "clip_ratio/low_mean": 0.08428190555423498, "clip_ratio/low_min": 0.08428190555423498, "clip_ratio/high_mean": 0.05872160289436579, "clip_ratio/high_max": 0.05872160289436579, "clip_ratio/region_mean": 0.14300350844860077, "reward_total_mean": 0.5244680047035217, "reward_meter_mean": 0.4777580499649048, "reward_meter_std": 0.36191046237945557, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9877919554710388, "reward_repeat_soft_std": 0.027078699320554733, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.22720351815223694, "reward_total_composite_mean": 0.5244680047035217, "reward_total_composite_std": 0.16301079094409943} {"timestamp_utc": "2026-04-13T09:14:53Z", "mode": "train", "global_step": 757, "epoch": 0.07604218985434455, "loss": 0.0962, "grad_norm": 18.8282527923584, "learning_rate": 7.709090909090909e-06, "num_tokens": 1337288.0, "completions/mean_length": 42.75, "completions/min_length": 36.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.75, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.3488878607749939, "rewards/meter/std": 0.3507755696773529, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9945725798606873, "rewards/repeat_soft/std": 0.005440371111035347, "rewards/judge_quality/mean": 0.5187499523162842, "rewards/judge_quality/std": 0.1799553632736206, "rewards/total_composite/mean": 0.45065516233444214, "rewards/total_composite/std": 0.100885771214962, "reward": 0.45065516233444214, "reward_std": 0.100885771214962, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20386451482772827, "sampling/sampling_logp_difference/max": 1.976332426071167, "sampling/importance_sampling_ratio/min": 0.13857655227184296, "sampling/importance_sampling_ratio/mean": 0.9969514608383179, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8613913357257843, "clip_ratio/low_mean": 0.11712655983865261, "clip_ratio/low_min": 0.11712655983865261, "clip_ratio/high_mean": 0.07038371358066797, "clip_ratio/high_max": 0.07038371358066797, "clip_ratio/region_mean": 0.18751027341932058, "reward_total_mean": 0.45065516233444214, "reward_meter_mean": 0.3488878607749939, "reward_meter_std": 0.3507755696773529, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9945725798606873, "reward_repeat_soft_std": 0.005440371111035347, "reward_judge_quality_mean": 0.5187499523162842, "reward_judge_quality_std": 0.1799553632736206, "reward_total_composite_mean": 0.45065516233444214, "reward_total_composite_std": 0.100885771214962} {"timestamp_utc": "2026-04-13T09:14:59Z", "mode": "train", "global_step": 758, "epoch": 0.07614264188849824, "loss": 0.0404, "grad_norm": 17.19386100769043, "learning_rate": 7.706060606060606e-06, "num_tokens": 1338817.0, "completions/mean_length": 46.125, "completions/min_length": 38.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.125, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.5067713260650635, "rewards/meter/std": 0.3364039361476898, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9228519201278687, "rewards/repeat_soft/std": 0.13501907885074615, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.4796302914619446, "rewards/total_composite/std": 0.08091279119253159, "reward": 0.4796302914619446, "reward_std": 0.08091279119253159, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19849666953086853, "sampling/sampling_logp_difference/max": 1.488032341003418, "sampling/importance_sampling_ratio/min": 0.225816547870636, "sampling/importance_sampling_ratio/mean": 1.0271035432815552, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.723602756857872, "clip_ratio/low_mean": 0.07782778888940811, "clip_ratio/low_min": 0.07782778888940811, "clip_ratio/high_mean": 0.0803694874048233, "clip_ratio/high_max": 0.0803694874048233, "clip_ratio/region_mean": 0.15819727629423141, "reward_total_mean": 0.4796302914619446, "reward_meter_mean": 0.5067713260650635, "reward_meter_std": 0.3364039361476898, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9228519201278687, "reward_repeat_soft_std": 0.13501907885074615, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.4796302914619446, "reward_total_composite_std": 0.08091279119253159} {"timestamp_utc": "2026-04-13T09:15:11Z", "mode": "train", "global_step": 759, "epoch": 0.07624309392265194, "loss": -0.184, "grad_norm": 3.580249309539795, "learning_rate": 7.703030303030304e-06, "num_tokens": 1341202.0, "completions/mean_length": 168.125, "completions/min_length": 91.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 119.00000762939453, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 138.0, "rewards/meter/mean": 0.5677632093429565, "rewards/meter/std": 0.3104856312274933, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.14880475401878357, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9539953470230103, "rewards/repeat_soft/std": 0.027996771037578583, "rewards/judge_quality/mean": 0.48625001311302185, "rewards/judge_quality/std": 0.23046152293682098, "rewards/total_composite/mean": 0.49184703826904297, "rewards/total_composite/std": 0.22149476408958435, "reward": 0.49184703826904297, "reward_std": 0.22149474918842316, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17595018446445465, "sampling/sampling_logp_difference/max": 1.7975578308105469, "sampling/importance_sampling_ratio/min": 0.16570307314395905, "sampling/importance_sampling_ratio/mean": 1.0275319814682007, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1057077646255493, "clip_ratio/low_mean": 0.041847825050354004, "clip_ratio/low_min": 0.041847825050354004, "clip_ratio/high_mean": 0.09729828126728535, "clip_ratio/high_max": 0.09729828126728535, "clip_ratio/region_mean": 0.13914610631763935, "reward_total_mean": 0.49184703826904297, "reward_meter_mean": 0.5677632093429565, "reward_meter_std": 0.3104856312274933, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.14880475401878357, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9539953470230103, "reward_repeat_soft_std": 0.027996771037578583, "reward_judge_quality_mean": 0.48625001311302185, "reward_judge_quality_std": 0.23046152293682098, "reward_total_composite_mean": 0.49184703826904297, "reward_total_composite_std": 0.22149476408958435} {"timestamp_utc": "2026-04-13T09:15:17Z", "mode": "train", "global_step": 760, "epoch": 0.07634354595680562, "loss": 0.0051, "grad_norm": 14.544023513793945, "learning_rate": 7.7e-06, "num_tokens": 1342872.0, "completions/mean_length": 45.75, "completions/min_length": 38.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.75, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.8717202544212341, "rewards/meter/std": 0.3210260570049286, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9277708530426025, "rewards/repeat_soft/std": 0.043750494718551636, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.10507651418447495, "rewards/total_composite/mean": 0.601246178150177, "rewards/total_composite/std": 0.11626143753528595, "reward": 0.601246178150177, "reward_std": 0.11626144498586655, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14291509985923767, "sampling/sampling_logp_difference/max": 1.8930234909057617, "sampling/importance_sampling_ratio/min": 0.15061573684215546, "sampling/importance_sampling_ratio/mean": 1.024094820022583, "sampling/importance_sampling_ratio/max": 1.982306957244873, "entropy": 0.9761926680803299, "clip_ratio/low_mean": 0.052287583239376545, "clip_ratio/low_min": 0.052287583239376545, "clip_ratio/high_mean": 0.09623707085847855, "clip_ratio/high_max": 0.09623707085847855, "clip_ratio/region_mean": 0.1485246540978551, "reward_total_mean": 0.601246178150177, "reward_meter_mean": 0.8717202544212341, "reward_meter_std": 0.3210260570049286, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9277708530426025, "reward_repeat_soft_std": 0.043750494718551636, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.10507651418447495, "reward_total_composite_mean": 0.601246178150177, "reward_total_composite_std": 0.11626143753528595} {"timestamp_utc": "2026-04-13T09:15:29Z", "mode": "train", "global_step": 761, "epoch": 0.07644399799095931, "loss": -0.2033, "grad_norm": 2.647010087966919, "learning_rate": 7.696969696969696e-06, "num_tokens": 1345400.0, "completions/mean_length": 188.0, "completions/min_length": 124.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 141.71429443359375, "completions/min_terminated_length": 124.0, "completions/max_terminated_length": 173.0, "rewards/meter/mean": 0.8462285399436951, "rewards/meter/std": 0.2651536762714386, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.12400396168231964, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9577072262763977, "rewards/repeat_soft/std": 0.024801496416330338, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.5141828060150146, "rewards/total_composite/std": 0.21469427645206451, "reward": 0.5141828060150146, "reward_std": 0.21469426155090332, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1781354397535324, "sampling/sampling_logp_difference/max": 1.9977011680603027, "sampling/importance_sampling_ratio/min": 0.1356467604637146, "sampling/importance_sampling_ratio/mean": 1.0326799154281616, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3835592418909073, "clip_ratio/low_mean": 0.018786126747727394, "clip_ratio/low_min": 0.018786126747727394, "clip_ratio/high_mean": 0.11646628007292747, "clip_ratio/high_max": 0.11646628007292747, "clip_ratio/region_mean": 0.13525240682065487, "reward_total_mean": 0.5141828060150146, "reward_meter_mean": 0.8462285399436951, "reward_meter_std": 0.2651536762714386, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.12400396168231964, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9577072262763977, "reward_repeat_soft_std": 0.024801496416330338, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.5141828060150146, "reward_total_composite_std": 0.21469427645206451} {"timestamp_utc": "2026-04-13T09:15:36Z", "mode": "train", "global_step": 762, "epoch": 0.07654445002511301, "loss": 0.1127, "grad_norm": 12.662678718566895, "learning_rate": 7.693939393939395e-06, "num_tokens": 1347943.0, "completions/mean_length": 107.875, "completions/min_length": 84.0, "completions/max_length": 160.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.875, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 160.0, "rewards/meter/mean": 0.842235267162323, "rewards/meter/std": 0.20652878284454346, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.10690449178218842, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9619722366333008, "rewards/repeat_soft/std": 0.032441169023513794, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.13695022463798523, "rewards/total_composite/mean": 0.44063663482666016, "rewards/total_composite/std": 0.2853558659553528, "reward": 0.44063663482666016, "reward_std": 0.28535589575767517, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17923206090927124, "sampling/sampling_logp_difference/max": 2.0224971771240234, "sampling/importance_sampling_ratio/min": 0.13232462108135223, "sampling/importance_sampling_ratio/mean": 0.9922921657562256, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0965612456202507, "clip_ratio/low_mean": 0.02926973532885313, "clip_ratio/low_min": 0.02926973532885313, "clip_ratio/high_mean": 0.1326370146125555, "clip_ratio/high_max": 0.1326370146125555, "clip_ratio/region_mean": 0.16190674994140863, "reward_total_mean": 0.44063663482666016, "reward_meter_mean": 0.842235267162323, "reward_meter_std": 0.20652878284454346, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.10690449178218842, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9619722366333008, "reward_repeat_soft_std": 0.032441169023513794, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.13695022463798523, "reward_total_composite_mean": 0.44063663482666016, "reward_total_composite_std": 0.2853558659553528} {"timestamp_utc": "2026-04-13T09:15:47Z", "mode": "train", "global_step": 763, "epoch": 0.0766449020592667, "loss": -0.1103, "grad_norm": 3.1770293712615967, "learning_rate": 7.690909090909091e-06, "num_tokens": 1349696.0, "completions/mean_length": 123.125, "completions/min_length": 65.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 67.5714340209961, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.3883039951324463, "rewards/meter/std": 0.3838312327861786, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.2357022762298584, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9750188589096069, "rewards/repeat_soft/std": 0.023989897221326828, "rewards/judge_quality/mean": 0.5987499952316284, "rewards/judge_quality/std": 0.28316769003868103, "rewards/total_composite/mean": 0.4377855360507965, "rewards/total_composite/std": 0.2552489638328552, "reward": 0.4377855360507965, "reward_std": 0.2552489638328552, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13883692026138306, "sampling/sampling_logp_difference/max": 3.3449044227600098, "sampling/importance_sampling_ratio/min": 0.03526358678936958, "sampling/importance_sampling_ratio/mean": 1.0103102922439575, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7316030263900757, "clip_ratio/low_mean": 0.057411215268075466, "clip_ratio/low_min": 0.057411215268075466, "clip_ratio/high_mean": 0.03219697065651417, "clip_ratio/high_max": 0.03219697065651417, "clip_ratio/region_mean": 0.08960818592458963, "reward_total_mean": 0.4377855360507965, "reward_meter_mean": 0.3883039951324463, "reward_meter_std": 0.3838312327861786, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.2357022762298584, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9750188589096069, "reward_repeat_soft_std": 0.023989897221326828, "reward_judge_quality_mean": 0.5987499952316284, "reward_judge_quality_std": 0.28316769003868103, "reward_total_composite_mean": 0.4377855360507965, "reward_total_composite_std": 0.2552489638328552} {"timestamp_utc": "2026-04-13T09:15:54Z", "mode": "train", "global_step": 764, "epoch": 0.0767453540934204, "loss": 0.028, "grad_norm": 14.50843334197998, "learning_rate": 7.687878787878788e-06, "num_tokens": 1351209.0, "completions/mean_length": 44.125, "completions/min_length": 42.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.125, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9126839637756348, "rewards/meter/std": 0.20989079773426056, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9592670202255249, "rewards/repeat_soft/std": 0.04412023723125458, "rewards/judge_quality/mean": 0.5824999809265137, "rewards/judge_quality/std": 0.23260943591594696, "rewards/total_composite/mean": 0.6975982189178467, "rewards/total_composite/std": 0.1767217367887497, "reward": 0.6975982189178467, "reward_std": 0.1767217367887497, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15735089778900146, "sampling/sampling_logp_difference/max": 1.154123306274414, "sampling/importance_sampling_ratio/min": 0.31533387303352356, "sampling/importance_sampling_ratio/mean": 1.0255959033966064, "sampling/importance_sampling_ratio/max": 1.9996552467346191, "entropy": 1.3363846838474274, "clip_ratio/low_mean": 0.08857548795640469, "clip_ratio/low_min": 0.08857548795640469, "clip_ratio/high_mean": 0.06425290182232857, "clip_ratio/high_max": 0.06425290182232857, "clip_ratio/region_mean": 0.15282838977873325, "reward_total_mean": 0.6975982189178467, "reward_meter_mean": 0.9126839637756348, "reward_meter_std": 0.20989079773426056, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9592670202255249, "reward_repeat_soft_std": 0.04412023723125458, "reward_judge_quality_mean": 0.5824999809265137, "reward_judge_quality_std": 0.23260943591594696, "reward_total_composite_mean": 0.6975982189178467, "reward_total_composite_std": 0.1767217367887497} {"timestamp_utc": "2026-04-13T09:16:05Z", "mode": "train", "global_step": 765, "epoch": 0.07684580612757408, "loss": -0.072, "grad_norm": 4.216747283935547, "learning_rate": 7.684848484848485e-06, "num_tokens": 1352574.0, "completions/mean_length": 85.625, "completions/min_length": 22.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 24.71428680419922, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.6552296876907349, "rewards/meter/std": 0.433675616979599, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9599957466125488, "rewards/repeat_soft/std": 0.0070829675532877445, "rewards/judge_quality/mean": 0.39625000953674316, "rewards/judge_quality/std": 0.14029940962791443, "rewards/total_composite/mean": 0.45586252212524414, "rewards/total_composite/std": 0.2198062390089035, "reward": 0.45586252212524414, "reward_std": 0.2198062390089035, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19094108045101166, "sampling/sampling_logp_difference/max": 1.128258228302002, "sampling/importance_sampling_ratio/min": 0.32359638810157776, "sampling/importance_sampling_ratio/mean": 1.0259119272232056, "sampling/importance_sampling_ratio/max": 1.927560567855835, "entropy": 1.506519839167595, "clip_ratio/low_mean": 0.06678511761128902, "clip_ratio/low_min": 0.06678511761128902, "clip_ratio/high_mean": 0.10454833135008812, "clip_ratio/high_max": 0.10454833135008812, "clip_ratio/region_mean": 0.17133344896137714, "reward_total_mean": 0.45586252212524414, "reward_meter_mean": 0.6552296876907349, "reward_meter_std": 0.433675616979599, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9599957466125488, "reward_repeat_soft_std": 0.0070829675532877445, "reward_judge_quality_mean": 0.39625000953674316, "reward_judge_quality_std": 0.14029940962791443, "reward_total_composite_mean": 0.45586252212524414, "reward_total_composite_std": 0.2198062390089035} {"timestamp_utc": "2026-04-13T09:16:11Z", "mode": "train", "global_step": 766, "epoch": 0.07694625816172777, "loss": 0.0643, "grad_norm": 13.96207332611084, "learning_rate": 7.681818181818183e-06, "num_tokens": 1354414.0, "completions/mean_length": 52.0, "completions/min_length": 46.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.0, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9453797340393066, "rewards/meter/std": 0.10096082091331482, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9896356463432312, "rewards/repeat_soft/std": 0.018915308639407158, "rewards/judge_quality/mean": 0.5824999809265137, "rewards/judge_quality/std": 0.23260943591594696, "rewards/total_composite/mean": 0.7106248140335083, "rewards/total_composite/std": 0.158878356218338, "reward": 0.7106248140335083, "reward_std": 0.15887834131717682, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15723644196987152, "sampling/sampling_logp_difference/max": 1.598418951034546, "sampling/importance_sampling_ratio/min": 0.2022159844636917, "sampling/importance_sampling_ratio/mean": 1.0136470794677734, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1226835399866104, "clip_ratio/low_mean": 0.09415237419307232, "clip_ratio/low_min": 0.09415237419307232, "clip_ratio/high_mean": 0.06614341773092747, "clip_ratio/high_max": 0.06614341773092747, "clip_ratio/region_mean": 0.1602957919239998, "reward_total_mean": 0.7106248140335083, "reward_meter_mean": 0.9453797340393066, "reward_meter_std": 0.10096082091331482, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9896356463432312, "reward_repeat_soft_std": 0.018915308639407158, "reward_judge_quality_mean": 0.5824999809265137, "reward_judge_quality_std": 0.23260943591594696, "reward_total_composite_mean": 0.7106248140335083, "reward_total_composite_std": 0.158878356218338} {"timestamp_utc": "2026-04-13T09:16:18Z", "mode": "train", "global_step": 767, "epoch": 0.07704671019588147, "loss": 0.0326, "grad_norm": 15.807914733886719, "learning_rate": 7.678787878787878e-06, "num_tokens": 1356408.0, "completions/mean_length": 80.25, "completions/min_length": 65.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.25, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.5978748798370361, "rewards/meter/std": 0.2955169975757599, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9931373596191406, "rewards/repeat_soft/std": 0.005969149060547352, "rewards/judge_quality/mean": 0.5824999809265137, "rewards/judge_quality/std": 0.23260943591594696, "rewards/total_composite/mean": 0.5544592142105103, "rewards/total_composite/std": 0.12938223779201508, "reward": 0.5544592142105103, "reward_std": 0.12938222289085388, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19241324067115784, "sampling/sampling_logp_difference/max": 2.6200027465820312, "sampling/importance_sampling_ratio/min": 0.07280266284942627, "sampling/importance_sampling_ratio/mean": 0.9966608285903931, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7311543300747871, "clip_ratio/low_mean": 0.11057438887655735, "clip_ratio/low_min": 0.11057438887655735, "clip_ratio/high_mean": 0.05676700547337532, "clip_ratio/high_max": 0.05676700547337532, "clip_ratio/region_mean": 0.16734139434993267, "reward_total_mean": 0.5544592142105103, "reward_meter_mean": 0.5978748798370361, "reward_meter_std": 0.2955169975757599, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9931373596191406, "reward_repeat_soft_std": 0.005969149060547352, "reward_judge_quality_mean": 0.5824999809265137, "reward_judge_quality_std": 0.23260943591594696, "reward_total_composite_mean": 0.5544592142105103, "reward_total_composite_std": 0.12938223779201508} {"timestamp_utc": "2026-04-13T09:16:24Z", "mode": "train", "global_step": 768, "epoch": 0.07714716223003516, "loss": 0.126, "grad_norm": 17.3958683013916, "learning_rate": 7.675757575757577e-06, "num_tokens": 1357981.0, "completions/mean_length": 27.625, "completions/min_length": 22.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.625, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.7603672742843628, "rewards/meter/std": 0.3146916329860687, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9957367777824402, "rewards/repeat_soft/std": 0.0060963560827076435, "rewards/judge_quality/mean": 0.7737500667572021, "rewards/judge_quality/std": 0.27458739280700684, "rewards/total_composite/mean": 0.728003203868866, "rewards/total_composite/std": 0.3442491888999939, "reward": 0.728003203868866, "reward_std": 0.3442491888999939, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1403573751449585, "sampling/sampling_logp_difference/max": 1.4121198654174805, "sampling/importance_sampling_ratio/min": 0.24362626671791077, "sampling/importance_sampling_ratio/mean": 1.0180941820144653, "sampling/importance_sampling_ratio/max": 1.9689271450042725, "entropy": 0.9881499111652374, "clip_ratio/low_mean": 0.03348214365541935, "clip_ratio/low_min": 0.03348214365541935, "clip_ratio/high_mean": 0.10506078135222197, "clip_ratio/high_max": 0.10506078135222197, "clip_ratio/region_mean": 0.13854292500764132, "reward_total_mean": 0.728003203868866, "reward_meter_mean": 0.7603672742843628, "reward_meter_std": 0.3146916329860687, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9957367777824402, "reward_repeat_soft_std": 0.0060963560827076435, "reward_judge_quality_mean": 0.7737500667572021, "reward_judge_quality_std": 0.27458739280700684, "reward_total_composite_mean": 0.728003203868866, "reward_total_composite_std": 0.3442491888999939} {"timestamp_utc": "2026-04-13T09:16:30Z", "mode": "train", "global_step": 769, "epoch": 0.07724761426418884, "loss": 0.0226, "grad_norm": 18.644556045532227, "learning_rate": 7.672727272727273e-06, "num_tokens": 1359397.0, "completions/mean_length": 25.0, "completions/min_length": 19.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.0, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.8413082957267761, "rewards/meter/std": 0.3381594717502594, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8590782880783081, "rewards/repeat_soft/std": 0.1899154782295227, "rewards/judge_quality/mean": 0.45625001192092896, "rewards/judge_quality/std": 0.21185827255249023, "rewards/total_composite/mean": 0.5824275016784668, "rewards/total_composite/std": 0.1867597997188568, "reward": 0.5824275016784668, "reward_std": 0.18675978481769562, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14683429896831512, "sampling/sampling_logp_difference/max": 2.0840606689453125, "sampling/importance_sampling_ratio/min": 0.12442393600940704, "sampling/importance_sampling_ratio/mean": 1.0149449110031128, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7047759927809238, "clip_ratio/low_mean": 0.04124658228829503, "clip_ratio/low_min": 0.04124658228829503, "clip_ratio/high_mean": 0.07379701640456915, "clip_ratio/high_max": 0.07379701640456915, "clip_ratio/region_mean": 0.11504359869286418, "reward_total_mean": 0.5824275016784668, "reward_meter_mean": 0.8413082957267761, "reward_meter_std": 0.3381594717502594, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8590782880783081, "reward_repeat_soft_std": 0.1899154782295227, "reward_judge_quality_mean": 0.45625001192092896, "reward_judge_quality_std": 0.21185827255249023, "reward_total_composite_mean": 0.5824275016784668, "reward_total_composite_std": 0.1867597997188568} {"timestamp_utc": "2026-04-13T09:16:41Z", "mode": "train", "global_step": 770, "epoch": 0.07734806629834254, "loss": -0.1785, "grad_norm": 3.450512170791626, "learning_rate": 7.66969696969697e-06, "num_tokens": 1361206.0, "completions/mean_length": 129.125, "completions/min_length": 65.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 74.42857360839844, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.9283363819122314, "rewards/meter/std": 0.13040593266487122, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9898082613945007, "rewards/repeat_soft/std": 0.010055318474769592, "rewards/judge_quality/mean": 0.5387499928474426, "rewards/judge_quality/std": 0.2517049014568329, "rewards/total_composite/mean": 0.587196946144104, "rewards/total_composite/std": 0.2623291611671448, "reward": 0.587196946144104, "reward_std": 0.2623291611671448, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16380071640014648, "sampling/sampling_logp_difference/max": 1.6352181434631348, "sampling/importance_sampling_ratio/min": 0.2790594696998596, "sampling/importance_sampling_ratio/mean": 1.0115994215011597, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0749797523021698, "clip_ratio/low_mean": 0.06256871297955513, "clip_ratio/low_min": 0.06256871297955513, "clip_ratio/high_mean": 0.08013118803501129, "clip_ratio/high_max": 0.08013118803501129, "clip_ratio/region_mean": 0.14269990101456642, "reward_total_mean": 0.587196946144104, "reward_meter_mean": 0.9283363819122314, "reward_meter_std": 0.13040593266487122, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9898082613945007, "reward_repeat_soft_std": 0.010055318474769592, "reward_judge_quality_mean": 0.5387499928474426, "reward_judge_quality_std": 0.2517049014568329, "reward_total_composite_mean": 0.587196946144104, "reward_total_composite_std": 0.2623291611671448} {"timestamp_utc": "2026-04-13T09:16:47Z", "mode": "train", "global_step": 771, "epoch": 0.07744851833249623, "loss": 0.0829, "grad_norm": 12.254379272460938, "learning_rate": 7.666666666666667e-06, "num_tokens": 1362865.0, "completions/mean_length": 57.375, "completions/min_length": 45.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.375, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9686471223831177, "rewards/meter/std": 0.045269280672073364, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9945254325866699, "rewards/repeat_soft/std": 0.008712784387171268, "rewards/judge_quality/mean": 0.48000001907348633, "rewards/judge_quality/std": 0.19071295857429504, "rewards/total_composite/mean": 0.6532793045043945, "rewards/total_composite/std": 0.12339124828577042, "reward": 0.6532793045043945, "reward_std": 0.12339124828577042, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1452702134847641, "sampling/sampling_logp_difference/max": 1.4859485626220703, "sampling/importance_sampling_ratio/min": 0.2262876033782959, "sampling/importance_sampling_ratio/mean": 1.0236133337020874, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0772370174527168, "clip_ratio/low_mean": 0.13615667633712292, "clip_ratio/low_min": 0.13615667633712292, "clip_ratio/high_mean": 0.01715686358511448, "clip_ratio/high_max": 0.01715686358511448, "clip_ratio/region_mean": 0.1533135399222374, "reward_total_mean": 0.6532793045043945, "reward_meter_mean": 0.9686471223831177, "reward_meter_std": 0.045269280672073364, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9945254325866699, "reward_repeat_soft_std": 0.008712784387171268, "reward_judge_quality_mean": 0.48000001907348633, "reward_judge_quality_std": 0.19071295857429504, "reward_total_composite_mean": 0.6532793045043945, "reward_total_composite_std": 0.12339124828577042} {"timestamp_utc": "2026-04-13T09:16:59Z", "mode": "train", "global_step": 772, "epoch": 0.07754897036664993, "loss": -0.0452, "grad_norm": 4.1646294593811035, "learning_rate": 7.663636363636364e-06, "num_tokens": 1364220.0, "completions/mean_length": 84.375, "completions/min_length": 15.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 23.285715103149414, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.45469433069229126, "rewards/meter/std": 0.47846660017967224, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9671874642372131, "rewards/repeat_soft/std": 0.013258260674774647, "rewards/judge_quality/mean": 0.39625000953674316, "rewards/judge_quality/std": 0.14029940962791443, "rewards/total_composite/mean": 0.4319379925727844, "rewards/total_composite/std": 0.2155667245388031, "reward": 0.4319379925727844, "reward_std": 0.2155667245388031, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22159840166568756, "sampling/sampling_logp_difference/max": 1.7078592777252197, "sampling/importance_sampling_ratio/min": 0.18125338852405548, "sampling/importance_sampling_ratio/mean": 1.0563842058181763, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5208516269922256, "clip_ratio/low_mean": 0.08533653989434242, "clip_ratio/low_min": 0.08533653989434242, "clip_ratio/high_mean": 0.08314394112676382, "clip_ratio/high_max": 0.08314394112676382, "clip_ratio/region_mean": 0.16848048102110624, "reward_total_mean": 0.4319379925727844, "reward_meter_mean": 0.45469433069229126, "reward_meter_std": 0.47846660017967224, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9671874642372131, "reward_repeat_soft_std": 0.013258260674774647, "reward_judge_quality_mean": 0.39625000953674316, "reward_judge_quality_std": 0.14029940962791443, "reward_total_composite_mean": 0.4319379925727844, "reward_total_composite_std": 0.2155667245388031} {"timestamp_utc": "2026-04-13T09:17:05Z", "mode": "train", "global_step": 773, "epoch": 0.07764942240080362, "loss": 0.0204, "grad_norm": 16.329193115234375, "learning_rate": 7.660606060606062e-06, "num_tokens": 1365963.0, "completions/mean_length": 46.875, "completions/min_length": 40.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.875, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.9151001572608948, "rewards/meter/std": 0.09497155994176865, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9848058223724365, "rewards/repeat_soft/std": 0.01939382031559944, "rewards/judge_quality/mean": 0.6225000023841858, "rewards/judge_quality/std": 0.24656209349632263, "rewards/total_composite/mean": 0.7250627279281616, "rewards/total_composite/std": 0.17324359714984894, "reward": 0.7250627279281616, "reward_std": 0.17324359714984894, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19107262790203094, "sampling/sampling_logp_difference/max": 2.556926727294922, "sampling/importance_sampling_ratio/min": 0.07754268497228622, "sampling/importance_sampling_ratio/mean": 1.00608491897583, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1131526306271553, "clip_ratio/low_mean": 0.07445781538262963, "clip_ratio/low_min": 0.07445781538262963, "clip_ratio/high_mean": 0.05299872066825628, "clip_ratio/high_max": 0.05299872066825628, "clip_ratio/region_mean": 0.12745653605088592, "reward_total_mean": 0.7250627279281616, "reward_meter_mean": 0.9151001572608948, "reward_meter_std": 0.09497155994176865, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9848058223724365, "reward_repeat_soft_std": 0.01939382031559944, "reward_judge_quality_mean": 0.6225000023841858, "reward_judge_quality_std": 0.24656209349632263, "reward_total_composite_mean": 0.7250627279281616, "reward_total_composite_std": 0.17324359714984894} {"timestamp_utc": "2026-04-13T09:17:11Z", "mode": "train", "global_step": 774, "epoch": 0.0777498744349573, "loss": 0.0682, "grad_norm": 11.950896263122559, "learning_rate": 7.657575757575757e-06, "num_tokens": 1367683.0, "completions/mean_length": 51.0, "completions/min_length": 38.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.0, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9761290550231934, "rewards/meter/std": 0.009273960255086422, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9809232950210571, "rewards/repeat_soft/std": 0.02996324747800827, "rewards/judge_quality/mean": 0.7400000095367432, "rewards/judge_quality/std": 0.214609295129776, "rewards/total_composite/mean": 0.8164946436882019, "rewards/total_composite/std": 0.1343652456998825, "reward": 0.8164946436882019, "reward_std": 0.1343652606010437, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1684437394142151, "sampling/sampling_logp_difference/max": 1.9768190383911133, "sampling/importance_sampling_ratio/min": 0.1385091245174408, "sampling/importance_sampling_ratio/mean": 1.0028111934661865, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1285023465752602, "clip_ratio/low_mean": 0.09706372953951359, "clip_ratio/low_min": 0.09706372953951359, "clip_ratio/high_mean": 0.0847039483487606, "clip_ratio/high_max": 0.0847039483487606, "clip_ratio/region_mean": 0.1817676778882742, "reward_total_mean": 0.8164946436882019, "reward_meter_mean": 0.9761290550231934, "reward_meter_std": 0.009273960255086422, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9809232950210571, "reward_repeat_soft_std": 0.02996324747800827, "reward_judge_quality_mean": 0.7400000095367432, "reward_judge_quality_std": 0.214609295129776, "reward_total_composite_mean": 0.8164946436882019, "reward_total_composite_std": 0.1343652456998825} {"timestamp_utc": "2026-04-13T09:17:24Z", "mode": "train", "global_step": 775, "epoch": 0.077850326469111, "loss": -0.077, "grad_norm": 5.5654168128967285, "learning_rate": 7.654545454545456e-06, "num_tokens": 1369386.0, "completions/mean_length": 110.875, "completions/min_length": 47.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 53.57143020629883, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.7354999780654907, "rewards/meter/std": 0.42168018221855164, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9402517080307007, "rewards/repeat_soft/std": 0.06503720581531525, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.5412236452102661, "rewards/total_composite/std": 0.11183484643697739, "reward": 0.5412236452102661, "reward_std": 0.11183483898639679, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.187318354845047, "sampling/sampling_logp_difference/max": 1.4507150650024414, "sampling/importance_sampling_ratio/min": 0.2344026267528534, "sampling/importance_sampling_ratio/mean": 1.0134236812591553, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1164140701293945, "clip_ratio/low_mean": 0.02500000037252903, "clip_ratio/low_min": 0.02500000037252903, "clip_ratio/high_mean": 0.12107518129050732, "clip_ratio/high_max": 0.12107518129050732, "clip_ratio/region_mean": 0.14607518166303635, "reward_total_mean": 0.5412236452102661, "reward_meter_mean": 0.7354999780654907, "reward_meter_std": 0.42168018221855164, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9402517080307007, "reward_repeat_soft_std": 0.06503720581531525, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.5412236452102661, "reward_total_composite_std": 0.11183484643697739} {"timestamp_utc": "2026-04-13T09:17:35Z", "mode": "train", "global_step": 776, "epoch": 0.07795077850326469, "loss": -0.1016, "grad_norm": 3.9890058040618896, "learning_rate": 7.651515151515152e-06, "num_tokens": 1370871.0, "completions/mean_length": 105.625, "completions/min_length": 41.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 47.57143020629883, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.6314957141876221, "rewards/meter/std": 0.37059882283210754, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9793222546577454, "rewards/repeat_soft/std": 0.03290608525276184, "rewards/judge_quality/mean": 0.48875001072883606, "rewards/judge_quality/std": 0.25147777795791626, "rewards/total_composite/mean": 0.5274893045425415, "rewards/total_composite/std": 0.2628191411495209, "reward": 0.5274893045425415, "reward_std": 0.2628191411495209, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15183840692043304, "sampling/sampling_logp_difference/max": 2.443131446838379, "sampling/importance_sampling_ratio/min": 0.08688833564519882, "sampling/importance_sampling_ratio/mean": 1.0251280069351196, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8372940421104431, "clip_ratio/low_mean": 0.04993386194109917, "clip_ratio/low_min": 0.04993386194109917, "clip_ratio/high_mean": 0.1101582683622837, "clip_ratio/high_max": 0.1101582683622837, "clip_ratio/region_mean": 0.16009213030338287, "reward_total_mean": 0.5274893045425415, "reward_meter_mean": 0.6314957141876221, "reward_meter_std": 0.37059882283210754, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9793222546577454, "reward_repeat_soft_std": 0.03290608525276184, "reward_judge_quality_mean": 0.48875001072883606, "reward_judge_quality_std": 0.25147777795791626, "reward_total_composite_mean": 0.5274893045425415, "reward_total_composite_std": 0.2628191411495209} {"timestamp_utc": "2026-04-13T09:17:42Z", "mode": "train", "global_step": 777, "epoch": 0.07805123053741839, "loss": -0.0499, "grad_norm": 19.725584030151367, "learning_rate": 7.648484848484849e-06, "num_tokens": 1372369.0, "completions/mean_length": 24.25, "completions/min_length": 20.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.25, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.8873765468597412, "rewards/meter/std": 0.20461690425872803, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9608622789382935, "rewards/repeat_soft/std": 0.0046321069821715355, "rewards/judge_quality/mean": 0.46000000834465027, "rewards/judge_quality/std": 0.21138995885849, "rewards/total_composite/mean": 0.6119965314865112, "rewards/total_composite/std": 0.15210434794425964, "reward": 0.6119965314865112, "reward_std": 0.15210434794425964, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15863265097141266, "sampling/sampling_logp_difference/max": 1.2769782543182373, "sampling/importance_sampling_ratio/min": 0.3066890239715576, "sampling/importance_sampling_ratio/mean": 1.0200525522232056, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.434073582291603, "clip_ratio/low_mean": 0.048363096080720425, "clip_ratio/low_min": 0.048363096080720425, "clip_ratio/high_mean": 0.10358843859285116, "clip_ratio/high_max": 0.10358843859285116, "clip_ratio/region_mean": 0.1519515346735716, "reward_total_mean": 0.6119965314865112, "reward_meter_mean": 0.8873765468597412, "reward_meter_std": 0.20461690425872803, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9608622789382935, "reward_repeat_soft_std": 0.0046321069821715355, "reward_judge_quality_mean": 0.46000000834465027, "reward_judge_quality_std": 0.21138995885849, "reward_total_composite_mean": 0.6119965314865112, "reward_total_composite_std": 0.15210434794425964} {"timestamp_utc": "2026-04-13T09:17:49Z", "mode": "train", "global_step": 778, "epoch": 0.07815168257157207, "loss": -0.0521, "grad_norm": 15.71272087097168, "learning_rate": 7.645454545454546e-06, "num_tokens": 1374322.0, "completions/mean_length": 64.125, "completions/min_length": 42.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.125, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.5612162351608276, "rewards/meter/std": 0.3150908052921295, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9959571957588196, "rewards/repeat_soft/std": 0.007008410524576902, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.10507651418447495, "rewards/total_composite/mean": 0.513142466545105, "rewards/total_composite/std": 0.11905184388160706, "reward": 0.513142466545105, "reward_std": 0.11905186623334885, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20146779716014862, "sampling/sampling_logp_difference/max": 1.9037284851074219, "sampling/importance_sampling_ratio/min": 0.1490119993686676, "sampling/importance_sampling_ratio/mean": 1.0142766237258911, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6660637855529785, "clip_ratio/low_mean": 0.08029035665094852, "clip_ratio/low_min": 0.08029035665094852, "clip_ratio/high_mean": 0.09320173598825932, "clip_ratio/high_max": 0.09320173598825932, "clip_ratio/region_mean": 0.17349209263920784, "reward_total_mean": 0.513142466545105, "reward_meter_mean": 0.5612162351608276, "reward_meter_std": 0.3150908052921295, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9959571957588196, "reward_repeat_soft_std": 0.007008410524576902, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.10507651418447495, "reward_total_composite_mean": 0.513142466545105, "reward_total_composite_std": 0.11905184388160706} {"timestamp_utc": "2026-04-13T09:17:55Z", "mode": "train", "global_step": 779, "epoch": 0.07825213460572576, "loss": 0.0016, "grad_norm": 13.117396354675293, "learning_rate": 7.642424242424244e-06, "num_tokens": 1376027.0, "completions/mean_length": 45.125, "completions/min_length": 33.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.125, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.3465504050254822, "rewards/meter/std": 0.26037847995758057, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9035850763320923, "rewards/repeat_soft/std": 0.11488734185695648, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.4339674711227417, "rewards/total_composite/std": 0.08193355053663254, "reward": 0.4339674711227417, "reward_std": 0.08193355798721313, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15073300898075104, "sampling/sampling_logp_difference/max": 1.0328598022460938, "sampling/importance_sampling_ratio/min": 0.35598745942115784, "sampling/importance_sampling_ratio/mean": 0.9965004920959473, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9764309599995613, "clip_ratio/low_mean": 0.08635712508112192, "clip_ratio/low_min": 0.08635712508112192, "clip_ratio/high_mean": 0.04932301864027977, "clip_ratio/high_max": 0.04932301864027977, "clip_ratio/region_mean": 0.1356801437214017, "reward_total_mean": 0.4339674711227417, "reward_meter_mean": 0.3465504050254822, "reward_meter_std": 0.26037847995758057, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9035850763320923, "reward_repeat_soft_std": 0.11488734185695648, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.4339674711227417, "reward_total_composite_std": 0.08193355053663254} {"timestamp_utc": "2026-04-13T09:18:06Z", "mode": "train", "global_step": 780, "epoch": 0.07835258663987946, "loss": -0.1464, "grad_norm": 2.2677340507507324, "learning_rate": 7.639393939393939e-06, "num_tokens": 1377876.0, "completions/mean_length": 249.125, "completions/min_length": 78.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 91.4000015258789, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.6196017265319824, "rewards/meter/std": 0.33285483717918396, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.3471825420856476, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.7700977325439453, "rewards/repeat_soft/std": 0.1705545336008072, "rewards/judge_quality/mean": 0.3999999761581421, "rewards/judge_quality/std": 0.36781206727027893, "rewards/total_composite/mean": 0.3964031934738159, "rewards/total_composite/std": 0.3589183986186981, "reward": 0.3964031934738159, "reward_std": 0.35891836881637573, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12576289474964142, "sampling/sampling_logp_difference/max": 1.7001616954803467, "sampling/importance_sampling_ratio/min": 0.1826539784669876, "sampling/importance_sampling_ratio/mean": 1.0203227996826172, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5451840795576572, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.06860982486978173, "clip_ratio/high_max": 0.06860982486978173, "clip_ratio/region_mean": 0.06860982486978173, "reward_total_mean": 0.3964031934738159, "reward_meter_mean": 0.6196017265319824, "reward_meter_std": 0.33285483717918396, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.3471825420856476, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.7700977325439453, "reward_repeat_soft_std": 0.1705545336008072, "reward_judge_quality_mean": 0.3999999761581421, "reward_judge_quality_std": 0.36781206727027893, "reward_total_composite_mean": 0.3964031934738159, "reward_total_composite_std": 0.3589183986186981} {"timestamp_utc": "2026-04-13T09:18:18Z", "mode": "train", "global_step": 781, "epoch": 0.07845303867403315, "loss": -0.1202, "grad_norm": 3.2003910541534424, "learning_rate": 7.636363636363638e-06, "num_tokens": 1379420.0, "completions/mean_length": 107.0, "completions/min_length": 45.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 49.142860412597656, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.3380328416824341, "rewards/meter/std": 0.3630189299583435, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9931943416595459, "rewards/repeat_soft/std": 0.01353566162288189, "rewards/judge_quality/mean": 0.5212500095367432, "rewards/judge_quality/std": 0.2591986358165741, "rewards/total_composite/mean": 0.37957361340522766, "rewards/total_composite/std": 0.17293740808963776, "reward": 0.37957361340522766, "reward_std": 0.17293739318847656, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15182960033416748, "sampling/sampling_logp_difference/max": 1.8792412281036377, "sampling/importance_sampling_ratio/min": 0.15270593762397766, "sampling/importance_sampling_ratio/mean": 1.0260132551193237, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9700796380639076, "clip_ratio/low_mean": 0.060232844203710556, "clip_ratio/low_min": 0.060232844203710556, "clip_ratio/high_mean": 0.0648200111463666, "clip_ratio/high_max": 0.0648200111463666, "clip_ratio/region_mean": 0.12505285535007715, "reward_total_mean": 0.37957361340522766, "reward_meter_mean": 0.3380328416824341, "reward_meter_std": 0.3630189299583435, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9931943416595459, "reward_repeat_soft_std": 0.01353566162288189, "reward_judge_quality_mean": 0.5212500095367432, "reward_judge_quality_std": 0.2591986358165741, "reward_total_composite_mean": 0.37957361340522766, "reward_total_composite_std": 0.17293740808963776} {"timestamp_utc": "2026-04-13T09:18:29Z", "mode": "train", "global_step": 782, "epoch": 0.07855349070818685, "loss": -0.0638, "grad_norm": 3.63277530670166, "learning_rate": 7.633333333333334e-06, "num_tokens": 1380808.0, "completions/mean_length": 87.5, "completions/min_length": 23.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 26.85714340209961, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.5601080656051636, "rewards/meter/std": 0.4184094965457916, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9671874642372131, "rewards/repeat_soft/std": 0.013258260674774647, "rewards/judge_quality/mean": 0.6349999904632568, "rewards/judge_quality/std": 0.33161941170692444, "rewards/total_composite/mean": 0.5623338222503662, "rewards/total_composite/std": 0.2842698395252228, "reward": 0.5623338222503662, "reward_std": 0.2842698395252228, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17743803560733795, "sampling/sampling_logp_difference/max": 1.9913415908813477, "sampling/importance_sampling_ratio/min": 0.1365121603012085, "sampling/importance_sampling_ratio/mean": 1.0241336822509766, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8500548712909222, "clip_ratio/low_mean": 0.04614695254713297, "clip_ratio/low_min": 0.04614695254713297, "clip_ratio/high_mean": 0.08065619971603155, "clip_ratio/high_max": 0.08065619971603155, "clip_ratio/region_mean": 0.12680315226316452, "reward_total_mean": 0.5623338222503662, "reward_meter_mean": 0.5601080656051636, "reward_meter_std": 0.4184094965457916, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9671874642372131, "reward_repeat_soft_std": 0.013258260674774647, "reward_judge_quality_mean": 0.6349999904632568, "reward_judge_quality_std": 0.33161941170692444, "reward_total_composite_mean": 0.5623338222503662, "reward_total_composite_std": 0.2842698395252228} {"timestamp_utc": "2026-04-13T09:18:40Z", "mode": "train", "global_step": 783, "epoch": 0.07865394274234053, "loss": -0.1072, "grad_norm": 1.7008535861968994, "learning_rate": 7.630303030303031e-06, "num_tokens": 1382436.0, "completions/mean_length": 164.5, "completions/min_length": 39.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 48.66666793823242, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9663210511207581, "rewards/meter/std": 0.04540238901972771, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9975972771644592, "rewards/repeat_soft/std": 0.004779578186571598, "rewards/judge_quality/mean": 0.4137499928474426, "rewards/judge_quality/std": 0.2602711617946625, "rewards/total_composite/mean": 0.5138510465621948, "rewards/total_composite/std": 0.3300139904022217, "reward": 0.5138510465621948, "reward_std": 0.3300139904022217, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16095556318759918, "sampling/sampling_logp_difference/max": 2.0277771949768066, "sampling/importance_sampling_ratio/min": 0.13162778317928314, "sampling/importance_sampling_ratio/mean": 0.9975730180740356, "sampling/importance_sampling_ratio/max": 1.8377701044082642, "entropy": 1.0407484769821167, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.12459279596805573, "clip_ratio/high_max": 0.12459279596805573, "clip_ratio/region_mean": 0.12459279596805573, "reward_total_mean": 0.5138510465621948, "reward_meter_mean": 0.9663210511207581, "reward_meter_std": 0.04540238901972771, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9975972771644592, "reward_repeat_soft_std": 0.004779578186571598, "reward_judge_quality_mean": 0.4137499928474426, "reward_judge_quality_std": 0.2602711617946625, "reward_total_composite_mean": 0.5138510465621948, "reward_total_composite_std": 0.3300139904022217} {"timestamp_utc": "2026-04-13T09:18:51Z", "mode": "train", "global_step": 784, "epoch": 0.07875439477649422, "loss": -0.157, "grad_norm": 4.411523342132568, "learning_rate": 7.627272727272727e-06, "num_tokens": 1384468.0, "completions/mean_length": 134.0, "completions/min_length": 65.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 80.0, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 92.0, "rewards/meter/mean": 0.8589165806770325, "rewards/meter/std": 0.26121780276298523, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9155592918395996, "rewards/repeat_soft/std": 0.04644685238599777, "rewards/judge_quality/mean": 0.6862499713897705, "rewards/judge_quality/std": 0.34221702814102173, "rewards/total_composite/mean": 0.7048117518424988, "rewards/total_composite/std": 0.34871798753738403, "reward": 0.7048117518424988, "reward_std": 0.34871798753738403, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17850680649280548, "sampling/sampling_logp_difference/max": 1.966983675956726, "sampling/importance_sampling_ratio/min": 0.13987812399864197, "sampling/importance_sampling_ratio/mean": 1.0244739055633545, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.102909229695797, "clip_ratio/low_mean": 0.032152908854186535, "clip_ratio/low_min": 0.032152908854186535, "clip_ratio/high_mean": 0.09117072448134422, "clip_ratio/high_max": 0.09117072448134422, "clip_ratio/region_mean": 0.12332363333553076, "reward_total_mean": 0.7048117518424988, "reward_meter_mean": 0.8589165806770325, "reward_meter_std": 0.26121780276298523, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9155592918395996, "reward_repeat_soft_std": 0.04644685238599777, "reward_judge_quality_mean": 0.6862499713897705, "reward_judge_quality_std": 0.34221702814102173, "reward_total_composite_mean": 0.7048117518424988, "reward_total_composite_std": 0.34871798753738403} {"timestamp_utc": "2026-04-13T09:18:58Z", "mode": "train", "global_step": 785, "epoch": 0.07885484681064792, "loss": 0.0508, "grad_norm": 12.999509811401367, "learning_rate": 7.6242424242424254e-06, "num_tokens": 1386131.0, "completions/mean_length": 50.875, "completions/min_length": 43.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.875, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.8049522042274475, "rewards/meter/std": 0.3317568898200989, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9843548536300659, "rewards/repeat_soft/std": 0.029881445690989494, "rewards/judge_quality/mean": 0.5175000429153442, "rewards/judge_quality/std": 0.1348809152841568, "rewards/total_composite/mean": 0.6261520385742188, "rewards/total_composite/std": 0.14769762754440308, "reward": 0.6261520385742188, "reward_std": 0.14769762754440308, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17569702863693237, "sampling/sampling_logp_difference/max": 2.9056591987609863, "sampling/importance_sampling_ratio/min": 0.05471271649003029, "sampling/importance_sampling_ratio/mean": 1.011540174484253, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1616344675421715, "clip_ratio/low_mean": 0.0884865066036582, "clip_ratio/low_min": 0.0884865066036582, "clip_ratio/high_mean": 0.09933181293308735, "clip_ratio/high_max": 0.09933181293308735, "clip_ratio/region_mean": 0.18781831953674555, "reward_total_mean": 0.6261520385742188, "reward_meter_mean": 0.8049522042274475, "reward_meter_std": 0.3317568898200989, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9843548536300659, "reward_repeat_soft_std": 0.029881445690989494, "reward_judge_quality_mean": 0.5175000429153442, "reward_judge_quality_std": 0.1348809152841568, "reward_total_composite_mean": 0.6261520385742188, "reward_total_composite_std": 0.14769762754440308} {"timestamp_utc": "2026-04-13T09:19:04Z", "mode": "train", "global_step": 786, "epoch": 0.07895529884480161, "loss": 0.0085, "grad_norm": 21.665050506591797, "learning_rate": 7.621212121212122e-06, "num_tokens": 1387476.0, "completions/mean_length": 25.125, "completions/min_length": 22.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.125, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.6657093167304993, "rewards/meter/std": 0.3802475929260254, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9570496082305908, "rewards/repeat_soft/std": 0.015415864065289497, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.17136012017726898, "rewards/total_composite/mean": 0.5710169076919556, "rewards/total_composite/std": 0.1791563630104065, "reward": 0.5710169076919556, "reward_std": 0.1791563332080841, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14274726808071136, "sampling/sampling_logp_difference/max": 0.998407244682312, "sampling/importance_sampling_ratio/min": 0.3684658706188202, "sampling/importance_sampling_ratio/mean": 1.031922459602356, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0329799726605415, "clip_ratio/low_mean": 0.04865320026874542, "clip_ratio/low_min": 0.04865320026874542, "clip_ratio/high_mean": 0.05201863497495651, "clip_ratio/high_max": 0.05201863497495651, "clip_ratio/region_mean": 0.10067183524370193, "reward_total_mean": 0.5710169076919556, "reward_meter_mean": 0.6657093167304993, "reward_meter_std": 0.3802475929260254, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9570496082305908, "reward_repeat_soft_std": 0.015415864065289497, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.17136012017726898, "reward_total_composite_mean": 0.5710169076919556, "reward_total_composite_std": 0.1791563630104065} {"timestamp_utc": "2026-04-13T09:19:10Z", "mode": "train", "global_step": 787, "epoch": 0.0790557508789553, "loss": 0.0898, "grad_norm": 15.552066802978516, "learning_rate": 7.618181818181819e-06, "num_tokens": 1389106.0, "completions/mean_length": 40.75, "completions/min_length": 33.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.75, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.680222749710083, "rewards/meter/std": 0.3175264000892639, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9770693778991699, "rewards/repeat_soft/std": 0.020964395254850388, "rewards/judge_quality/mean": 0.6062500476837158, "rewards/judge_quality/std": 0.19390259683132172, "rewards/total_composite/mean": 0.6077351570129395, "rewards/total_composite/std": 0.13894174993038177, "reward": 0.6077351570129395, "reward_std": 0.1389417052268982, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1768689900636673, "sampling/sampling_logp_difference/max": 1.7020316123962402, "sampling/importance_sampling_ratio/min": 0.1823127567768097, "sampling/importance_sampling_ratio/mean": 1.0149352550506592, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.059500440955162, "clip_ratio/low_mean": 0.060787507332861423, "clip_ratio/low_min": 0.060787507332861423, "clip_ratio/high_mean": 0.08163997158408165, "clip_ratio/high_max": 0.08163997158408165, "clip_ratio/region_mean": 0.14242747891694307, "reward_total_mean": 0.6077351570129395, "reward_meter_mean": 0.680222749710083, "reward_meter_std": 0.3175264000892639, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9770693778991699, "reward_repeat_soft_std": 0.020964395254850388, "reward_judge_quality_mean": 0.6062500476837158, "reward_judge_quality_std": 0.19390259683132172, "reward_total_composite_mean": 0.6077351570129395, "reward_total_composite_std": 0.13894174993038177} {"timestamp_utc": "2026-04-13T09:19:17Z", "mode": "train", "global_step": 788, "epoch": 0.07915620291310899, "loss": -0.0161, "grad_norm": 14.044755935668945, "learning_rate": 7.6151515151515155e-06, "num_tokens": 1390843.0, "completions/mean_length": 47.125, "completions/min_length": 40.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.125, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.7587764263153076, "rewards/meter/std": 0.327594518661499, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9857375621795654, "rewards/repeat_soft/std": 0.01699240505695343, "rewards/judge_quality/mean": 0.5099999904632568, "rewards/judge_quality/std": 0.215406596660614, "rewards/total_composite/mean": 0.6128135919570923, "rewards/total_composite/std": 0.1883668303489685, "reward": 0.6128135919570923, "reward_std": 0.1883668452501297, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14258180558681488, "sampling/sampling_logp_difference/max": 1.9234857559204102, "sampling/importance_sampling_ratio/min": 0.14609681069850922, "sampling/importance_sampling_ratio/mean": 1.0270355939865112, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.011030912399292, "clip_ratio/low_mean": 0.1028314083814621, "clip_ratio/low_min": 0.1028314083814621, "clip_ratio/high_mean": 0.07616372779011726, "clip_ratio/high_max": 0.07616372779011726, "clip_ratio/region_mean": 0.17899513617157936, "reward_total_mean": 0.6128135919570923, "reward_meter_mean": 0.7587764263153076, "reward_meter_std": 0.327594518661499, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9857375621795654, "reward_repeat_soft_std": 0.01699240505695343, "reward_judge_quality_mean": 0.5099999904632568, "reward_judge_quality_std": 0.215406596660614, "reward_total_composite_mean": 0.6128135919570923, "reward_total_composite_std": 0.1883668303489685} {"timestamp_utc": "2026-04-13T09:19:23Z", "mode": "train", "global_step": 789, "epoch": 0.07925665494726268, "loss": 0.0352, "grad_norm": 16.33643913269043, "learning_rate": 7.612121212121213e-06, "num_tokens": 1392585.0, "completions/mean_length": 44.75, "completions/min_length": 36.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.75, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.9510666728019714, "rewards/meter/std": 0.05670829862356186, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9683701992034912, "rewards/repeat_soft/std": 0.0318145677447319, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.5726941823959351, "rewards/total_composite/std": 0.250593900680542, "reward": 0.5726941823959351, "reward_std": 0.250593900680542, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1539141684770584, "sampling/sampling_logp_difference/max": 2.2820472717285156, "sampling/importance_sampling_ratio/min": 0.1020750179886818, "sampling/importance_sampling_ratio/mean": 1.0005478858947754, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8115236088633537, "clip_ratio/low_mean": 0.025510204955935478, "clip_ratio/low_min": 0.025510204955935478, "clip_ratio/high_mean": 0.1278712498024106, "clip_ratio/high_max": 0.1278712498024106, "clip_ratio/region_mean": 0.15338145475834608, "reward_total_mean": 0.5726941823959351, "reward_meter_mean": 0.9510666728019714, "reward_meter_std": 0.05670829862356186, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9683701992034912, "reward_repeat_soft_std": 0.0318145677447319, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.5726941823959351, "reward_total_composite_std": 0.250593900680542} {"timestamp_utc": "2026-04-13T09:19:31Z", "mode": "train", "global_step": 790, "epoch": 0.07935710698141638, "loss": 0.0481, "grad_norm": 9.999359130859375, "learning_rate": 7.609090909090909e-06, "num_tokens": 1395365.0, "completions/mean_length": 127.5, "completions/min_length": 116.0, "completions/max_length": 157.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 127.5, "completions/min_terminated_length": 116.0, "completions/max_terminated_length": 157.0, "rewards/meter/mean": 0.6306518316268921, "rewards/meter/std": 0.3782559037208557, "rewards/count_adherence/mean": 0.8541666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9702376127243042, "rewards/repeat_soft/std": 0.031808041036129, "rewards/judge_quality/mean": 0.6575000286102295, "rewards/judge_quality/std": 0.2133910059928894, "rewards/total_composite/mean": 0.6025485396385193, "rewards/total_composite/std": 0.2148790806531906, "reward": 0.6025485396385193, "reward_std": 0.2148790806531906, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16975581645965576, "sampling/sampling_logp_difference/max": 2.3685126304626465, "sampling/importance_sampling_ratio/min": 0.09361986815929413, "sampling/importance_sampling_ratio/mean": 1.0210152864456177, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0747182965278625, "clip_ratio/low_mean": 0.05339346919208765, "clip_ratio/low_min": 0.05339346919208765, "clip_ratio/high_mean": 0.10562057327479124, "clip_ratio/high_max": 0.10562057327479124, "clip_ratio/region_mean": 0.1590140424668789, "reward_total_mean": 0.6025485396385193, "reward_meter_mean": 0.6306518316268921, "reward_meter_std": 0.3782559037208557, "reward_count_adherence_mean": 0.8541666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9702376127243042, "reward_repeat_soft_std": 0.031808041036129, "reward_judge_quality_mean": 0.6575000286102295, "reward_judge_quality_std": 0.2133910059928894, "reward_total_composite_mean": 0.6025485396385193, "reward_total_composite_std": 0.2148790806531906} {"timestamp_utc": "2026-04-13T09:19:42Z", "mode": "train", "global_step": 791, "epoch": 0.07945755901557007, "loss": -0.141, "grad_norm": 3.7419774532318115, "learning_rate": 7.606060606060606e-06, "num_tokens": 1397425.0, "completions/mean_length": 155.5, "completions/min_length": 91.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 104.5714340209961, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.6867519617080688, "rewards/meter/std": 0.4089537560939789, "rewards/count_adherence/mean": 0.84375, "rewards/count_adherence/std": 0.35197150707244873, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9493618011474609, "rewards/repeat_soft/std": 0.05099398270249367, "rewards/judge_quality/mean": 0.45249998569488525, "rewards/judge_quality/std": 0.21022097766399384, "rewards/total_composite/mean": 0.5299545526504517, "rewards/total_composite/std": 0.26412343978881836, "reward": 0.5299545526504517, "reward_std": 0.26412343978881836, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16901877522468567, "sampling/sampling_logp_difference/max": 1.9053401947021484, "sampling/importance_sampling_ratio/min": 0.17543934285640717, "sampling/importance_sampling_ratio/mean": 1.016891360282898, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9964922145009041, "clip_ratio/low_mean": 0.03585714288055897, "clip_ratio/low_min": 0.03585714288055897, "clip_ratio/high_mean": 0.08133437763899565, "clip_ratio/high_max": 0.08133437763899565, "clip_ratio/region_mean": 0.11719152051955462, "reward_total_mean": 0.5299545526504517, "reward_meter_mean": 0.6867519617080688, "reward_meter_std": 0.4089537560939789, "reward_count_adherence_mean": 0.84375, "reward_count_adherence_std": 0.35197150707244873, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9493618011474609, "reward_repeat_soft_std": 0.05099398270249367, "reward_judge_quality_mean": 0.45249998569488525, "reward_judge_quality_std": 0.21022097766399384, "reward_total_composite_mean": 0.5299545526504517, "reward_total_composite_std": 0.26412343978881836} {"timestamp_utc": "2026-04-13T09:19:48Z", "mode": "train", "global_step": 792, "epoch": 0.07955801104972375, "loss": 0.0258, "grad_norm": 15.24664306640625, "learning_rate": 7.603030303030303e-06, "num_tokens": 1398939.0, "completions/mean_length": 21.25, "completions/min_length": 18.0, "completions/max_length": 24.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.25, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 24.0, "rewards/meter/mean": 0.9517276287078857, "rewards/meter/std": 0.0785837322473526, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.961837112903595, "rewards/repeat_soft/std": 0.0018749026348814368, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.7555526494979858, "rewards/total_composite/std": 0.16212528944015503, "reward": 0.7555526494979858, "reward_std": 0.16212528944015503, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10954467952251434, "sampling/sampling_logp_difference/max": 0.9737614393234253, "sampling/importance_sampling_ratio/min": 0.3776598274707794, "sampling/importance_sampling_ratio/mean": 1.0057436227798462, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5773142501711845, "clip_ratio/low_mean": 0.08900091052055359, "clip_ratio/low_min": 0.08900091052055359, "clip_ratio/high_mean": 0.07013889029622078, "clip_ratio/high_max": 0.07013889029622078, "clip_ratio/region_mean": 0.15913980081677437, "reward_total_mean": 0.7555526494979858, "reward_meter_mean": 0.9517276287078857, "reward_meter_std": 0.0785837322473526, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.961837112903595, "reward_repeat_soft_std": 0.0018749026348814368, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.7555526494979858, "reward_total_composite_std": 0.16212528944015503} {"timestamp_utc": "2026-04-13T09:20:00Z", "mode": "train", "global_step": 793, "epoch": 0.07965846308387745, "loss": -0.0863, "grad_norm": 3.825547695159912, "learning_rate": 7.600000000000001e-06, "num_tokens": 1400431.0, "completions/mean_length": 98.5, "completions/min_length": 33.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 39.42857360839844, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.37778598070144653, "rewards/meter/std": 0.36534878611564636, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9773523807525635, "rewards/repeat_soft/std": 0.03622420132160187, "rewards/judge_quality/mean": 0.6062500476837158, "rewards/judge_quality/std": 0.32053250074386597, "rewards/total_composite/mean": 0.4723506569862366, "rewards/total_composite/std": 0.23831723630428314, "reward": 0.4723506569862366, "reward_std": 0.23831720650196075, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1502223163843155, "sampling/sampling_logp_difference/max": 2.3693289756774902, "sampling/importance_sampling_ratio/min": 0.09354346990585327, "sampling/importance_sampling_ratio/mean": 0.9848490357398987, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9237863123416901, "clip_ratio/low_mean": 0.038352273404598236, "clip_ratio/low_min": 0.038352273404598236, "clip_ratio/high_mean": 0.08816816285252571, "clip_ratio/high_max": 0.08816816285252571, "clip_ratio/region_mean": 0.12652043625712395, "reward_total_mean": 0.4723506569862366, "reward_meter_mean": 0.37778598070144653, "reward_meter_std": 0.36534878611564636, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9773523807525635, "reward_repeat_soft_std": 0.03622420132160187, "reward_judge_quality_mean": 0.6062500476837158, "reward_judge_quality_std": 0.32053250074386597, "reward_total_composite_mean": 0.4723506569862366, "reward_total_composite_std": 0.23831723630428314} {"timestamp_utc": "2026-04-13T09:20:13Z", "mode": "train", "global_step": 794, "epoch": 0.07975891511803114, "loss": 0.0329, "grad_norm": 9.547015190124512, "learning_rate": 7.596969696969697e-06, "num_tokens": 1402650.0, "completions/mean_length": 99.375, "completions/min_length": 91.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 99.375, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.9521958827972412, "rewards/meter/std": 0.07844063639640808, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9547011256217957, "rewards/repeat_soft/std": 0.025254275649785995, "rewards/judge_quality/mean": 0.7074999809265137, "rewards/judge_quality/std": 0.2474873960018158, "rewards/total_composite/mean": 0.7787088751792908, "rewards/total_composite/std": 0.15237843990325928, "reward": 0.7787088751792908, "reward_std": 0.15237843990325928, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1505635529756546, "sampling/sampling_logp_difference/max": 1.9828691482543945, "sampling/importance_sampling_ratio/min": 0.13767366111278534, "sampling/importance_sampling_ratio/mean": 1.0115715265274048, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0196355730295181, "clip_ratio/low_mean": 0.048223040997982025, "clip_ratio/low_min": 0.048223040997982025, "clip_ratio/high_mean": 0.0992923779413104, "clip_ratio/high_max": 0.0992923779413104, "clip_ratio/region_mean": 0.14751541893929243, "reward_total_mean": 0.7787088751792908, "reward_meter_mean": 0.9521958827972412, "reward_meter_std": 0.07844063639640808, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9547011256217957, "reward_repeat_soft_std": 0.025254275649785995, "reward_judge_quality_mean": 0.7074999809265137, "reward_judge_quality_std": 0.2474873960018158, "reward_total_composite_mean": 0.7787088751792908, "reward_total_composite_std": 0.15237843990325928} {"timestamp_utc": "2026-04-13T09:20:20Z", "mode": "train", "global_step": 795, "epoch": 0.07985936715218483, "loss": 0.0164, "grad_norm": 7.827341079711914, "learning_rate": 7.593939393939395e-06, "num_tokens": 1405076.0, "completions/mean_length": 114.25, "completions/min_length": 96.0, "completions/max_length": 137.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 114.25, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 137.0, "rewards/meter/mean": 0.9653751850128174, "rewards/meter/std": 0.04670365899801254, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9599523544311523, "rewards/repeat_soft/std": 0.03862236067652702, "rewards/judge_quality/mean": 0.6237499713897705, "rewards/judge_quality/std": 0.2232191562652588, "rewards/total_composite/mean": 0.738872766494751, "rewards/total_composite/std": 0.15536025166511536, "reward": 0.738872766494751, "reward_std": 0.15536023676395416, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16115161776542664, "sampling/sampling_logp_difference/max": 2.0109481811523438, "sampling/importance_sampling_ratio/min": 0.1338616907596588, "sampling/importance_sampling_ratio/mean": 1.0239852666854858, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1828294694423676, "clip_ratio/low_mean": 0.0740177072584629, "clip_ratio/low_min": 0.0740177072584629, "clip_ratio/high_mean": 0.07718428783118725, "clip_ratio/high_max": 0.07718428783118725, "clip_ratio/region_mean": 0.15120199508965015, "reward_total_mean": 0.738872766494751, "reward_meter_mean": 0.9653751850128174, "reward_meter_std": 0.04670365899801254, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9599523544311523, "reward_repeat_soft_std": 0.03862236067652702, "reward_judge_quality_mean": 0.6237499713897705, "reward_judge_quality_std": 0.2232191562652588, "reward_total_composite_mean": 0.738872766494751, "reward_total_composite_std": 0.15536025166511536} {"timestamp_utc": "2026-04-13T09:20:31Z", "mode": "train", "global_step": 796, "epoch": 0.07995981918633853, "loss": -0.076, "grad_norm": 3.3293709754943848, "learning_rate": 7.590909090909091e-06, "num_tokens": 1406630.0, "completions/mean_length": 102.25, "completions/min_length": 39.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 43.71428680419922, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.664996862411499, "rewards/meter/std": 0.3527509570121765, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.993262529373169, "rewards/repeat_soft/std": 0.00963639747351408, "rewards/judge_quality/mean": 0.668749988079071, "rewards/judge_quality/std": 0.3276948928833008, "rewards/total_composite/mean": 0.6403868198394775, "rewards/total_composite/std": 0.30043327808380127, "reward": 0.6403868198394775, "reward_std": 0.30043327808380127, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1478281021118164, "sampling/sampling_logp_difference/max": 1.687039852142334, "sampling/importance_sampling_ratio/min": 0.18506652116775513, "sampling/importance_sampling_ratio/mean": 0.9968125224113464, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7571103498339653, "clip_ratio/low_mean": 0.03758741356432438, "clip_ratio/low_min": 0.03758741356432438, "clip_ratio/high_mean": 0.09591932641342282, "clip_ratio/high_max": 0.09591932641342282, "clip_ratio/region_mean": 0.1335067399777472, "reward_total_mean": 0.6403868198394775, "reward_meter_mean": 0.664996862411499, "reward_meter_std": 0.3527509570121765, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.993262529373169, "reward_repeat_soft_std": 0.00963639747351408, "reward_judge_quality_mean": 0.668749988079071, "reward_judge_quality_std": 0.3276948928833008, "reward_total_composite_mean": 0.6403868198394775, "reward_total_composite_std": 0.30043327808380127} {"timestamp_utc": "2026-04-13T09:20:43Z", "mode": "train", "global_step": 797, "epoch": 0.08006027122049221, "loss": -0.1188, "grad_norm": 5.404216289520264, "learning_rate": 7.587878787878788e-06, "num_tokens": 1408726.0, "completions/mean_length": 154.0, "completions/min_length": 94.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 102.85714721679688, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 115.0, "rewards/meter/mean": 0.7925493717193604, "rewards/meter/std": 0.23417937755584717, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.8938276767730713, "rewards/repeat_soft/std": 0.07476700097322464, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.22222496569156647, "rewards/total_composite/mean": 0.4308021366596222, "rewards/total_composite/std": 0.2820538878440857, "reward": 0.4308021366596222, "reward_std": 0.2820538580417633, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1384115219116211, "sampling/sampling_logp_difference/max": 2.7043967247009277, "sampling/importance_sampling_ratio/min": 0.06691067665815353, "sampling/importance_sampling_ratio/mean": 0.9951090812683105, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7167848758399487, "clip_ratio/low_mean": 0.014627659693360329, "clip_ratio/low_min": 0.014627659693360329, "clip_ratio/high_mean": 0.11153530143201351, "clip_ratio/high_max": 0.11153530143201351, "clip_ratio/region_mean": 0.12616296112537384, "reward_total_mean": 0.4308021366596222, "reward_meter_mean": 0.7925493717193604, "reward_meter_std": 0.23417937755584717, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.8938276767730713, "reward_repeat_soft_std": 0.07476700097322464, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.22222496569156647, "reward_total_composite_mean": 0.4308021366596222, "reward_total_composite_std": 0.2820538878440857} {"timestamp_utc": "2026-04-13T09:20:49Z", "mode": "train", "global_step": 798, "epoch": 0.0801607232546459, "loss": 0.0659, "grad_norm": 13.294465065002441, "learning_rate": 7.584848484848486e-06, "num_tokens": 1410885.0, "completions/mean_length": 88.875, "completions/min_length": 66.0, "completions/max_length": 95.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 88.875, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.6034026145935059, "rewards/meter/std": 0.3710324764251709, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9907156229019165, "rewards/repeat_soft/std": 0.008669731207191944, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5133362412452698, "rewards/total_composite/std": 0.10131111741065979, "reward": 0.5133362412452698, "reward_std": 0.1013111099600792, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16643524169921875, "sampling/sampling_logp_difference/max": 8.26811408996582, "sampling/importance_sampling_ratio/min": 0.0002565686882007867, "sampling/importance_sampling_ratio/mean": 0.9939804077148438, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6501155868172646, "clip_ratio/low_mean": 0.07310330867767334, "clip_ratio/low_min": 0.07310330867767334, "clip_ratio/high_mean": 0.07407133094966412, "clip_ratio/high_max": 0.07407133094966412, "clip_ratio/region_mean": 0.14717463962733746, "reward_total_mean": 0.5133362412452698, "reward_meter_mean": 0.6034026145935059, "reward_meter_std": 0.3710324764251709, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9907156229019165, "reward_repeat_soft_std": 0.008669731207191944, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5133362412452698, "reward_total_composite_std": 0.10131111741065979} {"timestamp_utc": "2026-04-13T09:20:56Z", "mode": "train", "global_step": 799, "epoch": 0.0802611752887996, "loss": 0.0763, "grad_norm": 17.824810028076172, "learning_rate": 7.581818181818183e-06, "num_tokens": 1412684.0, "completions/mean_length": 57.875, "completions/min_length": 52.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.875, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.4198606312274933, "rewards/meter/std": 0.23079407215118408, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9929182529449463, "rewards/repeat_soft/std": 0.008453218266367912, "rewards/judge_quality/mean": 0.6612499952316284, "rewards/judge_quality/std": 0.2654612064361572, "rewards/total_composite/mean": 0.5252758264541626, "rewards/total_composite/std": 0.1283113658428192, "reward": 0.5252758264541626, "reward_std": 0.1283113658428192, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18238553404808044, "sampling/sampling_logp_difference/max": 2.830293655395508, "sampling/importance_sampling_ratio/min": 0.058995530009269714, "sampling/importance_sampling_ratio/mean": 1.0064098834991455, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.822184830904007, "clip_ratio/low_mean": 0.09742045029997826, "clip_ratio/low_min": 0.09742045029997826, "clip_ratio/high_mean": 0.05481120944023132, "clip_ratio/high_max": 0.05481120944023132, "clip_ratio/region_mean": 0.15223165974020958, "reward_total_mean": 0.5252758264541626, "reward_meter_mean": 0.4198606312274933, "reward_meter_std": 0.23079407215118408, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9929182529449463, "reward_repeat_soft_std": 0.008453218266367912, "reward_judge_quality_mean": 0.6612499952316284, "reward_judge_quality_std": 0.2654612064361572, "reward_total_composite_mean": 0.5252758264541626, "reward_total_composite_std": 0.1283113658428192} {"timestamp_utc": "2026-04-13T09:21:02Z", "mode": "train", "global_step": 800, "epoch": 0.0803616273229533, "loss": 0.0126, "grad_norm": 11.47986125946045, "learning_rate": 7.57878787878788e-06, "num_tokens": 1414466.0, "completions/mean_length": 50.75, "completions/min_length": 43.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.75, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.3981095552444458, "rewards/meter/std": 0.4597620666027069, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9975108504295349, "rewards/repeat_soft/std": 0.004614746663719416, "rewards/judge_quality/mean": 0.78125, "rewards/judge_quality/std": 0.2126995027065277, "rewards/total_composite/mean": 0.5501787662506104, "rewards/total_composite/std": 0.25034084916114807, "reward": 0.5501787662506104, "reward_std": 0.2503408193588257, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1491750329732895, "sampling/sampling_logp_difference/max": 1.6211563348770142, "sampling/importance_sampling_ratio/min": 0.19766999781131744, "sampling/importance_sampling_ratio/mean": 1.0138750076293945, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9692709296941757, "clip_ratio/low_mean": 0.09596501477062702, "clip_ratio/low_min": 0.09596501477062702, "clip_ratio/high_mean": 0.061666665598750114, "clip_ratio/high_max": 0.061666665598750114, "clip_ratio/region_mean": 0.15763168036937714, "reward_total_mean": 0.5501787662506104, "reward_meter_mean": 0.3981095552444458, "reward_meter_std": 0.4597620666027069, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9975108504295349, "reward_repeat_soft_std": 0.004614746663719416, "reward_judge_quality_mean": 0.78125, "reward_judge_quality_std": 0.2126995027065277, "reward_total_composite_mean": 0.5501787662506104, "reward_total_composite_std": 0.25034084916114807} {"timestamp_utc": "2026-04-13T09:21:55Z", "mode": "eval", "global_step": 800, "epoch": 0.0803616273229533, "eval_loss": NaN, "eval_runtime": 52.9423, "eval_samples_per_second": 1.511, "eval_steps_per_second": 0.189, "eval_num_tokens": 1414466.0, "eval_completions/mean_length": 96.1125, "eval_completions/min_length": 39.4, "eval_completions/max_length": 237.7, "eval_completions/clipped_ratio": 0.05, "eval_completions/mean_terminated_length": 74.84107208251953, "eval_completions/min_terminated_length": 39.4, "eval_completions/max_terminated_length": 128.6, "eval_rewards/meter/mean": 0.5917634904384613, "eval_rewards/meter/std": 0.34982125759124755, "eval_rewards/count_adherence/mean": 0.9554166615009307, "eval_rewards/count_adherence/std": 0.09762779586017131, "eval_rewards/hard_gate/mean": 0.9375, "eval_rewards/hard_gate/std": 0.15235702097415924, "eval_rewards/repeat_soft/mean": 0.9491789221763611, "eval_rewards/repeat_soft/std": 0.06461945287883282, "eval_rewards/judge_quality/mean": 0.5437499970197678, "eval_rewards/judge_quality/std": 0.21584835201501845, "eval_rewards/total_composite/mean": 0.5118943601846695, "eval_rewards/total_composite/std": 0.19045581221580504, "eval_reward": 0.5118943601846695, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.07544206082820892, "eval_sampling/sampling_logp_difference/max": 1.1279006004333496, "eval_sampling/importance_sampling_ratio/min": 0.33538752645254133, "eval_sampling/importance_sampling_ratio/mean": 1.02023948431015, "eval_sampling/importance_sampling_ratio/max": 1.4546257734298706, "eval_entropy": 0.865834218263626, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5118943601846695, "eval_reward_meter_mean": 0.5917634904384613, "eval_reward_meter_std": 0.34982125759124755, "eval_reward_count_adherence_mean": 0.9554166615009307, "eval_reward_count_adherence_std": 0.09762779586017131, "eval_reward_hard_gate_mean": 0.9375, "eval_reward_hard_gate_std": 0.15235702097415924, "eval_reward_repeat_soft_mean": 0.9491789221763611, "eval_reward_repeat_soft_std": 0.06461945287883282, "eval_reward_judge_quality_mean": 0.5437499970197678, "eval_reward_judge_quality_std": 0.21584835201501845, "eval_reward_total_composite_mean": 0.5118943601846695, "eval_reward_total_composite_std": 0.19045581221580504} {"timestamp_utc": "2026-04-13T09:22:04Z", "mode": "train", "global_step": 801, "epoch": 0.08046207935710697, "loss": 0.0383, "grad_norm": 14.906476974487305, "learning_rate": 7.5757575757575764e-06, "num_tokens": 1416082.0, "completions/mean_length": 42.0, "completions/min_length": 38.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.0, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.29853975772857666, "rewards/meter/std": 0.3955845832824707, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9885627627372742, "rewards/repeat_soft/std": 0.010870145633816719, "rewards/judge_quality/mean": 0.6575000286102295, "rewards/judge_quality/std": 0.2133910059928894, "rewards/total_composite/mean": 0.44581830501556396, "rewards/total_composite/std": 0.10586800426244736, "reward": 0.44581830501556396, "reward_std": 0.10586800426244736, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1568552553653717, "sampling/sampling_logp_difference/max": 2.663341999053955, "sampling/importance_sampling_ratio/min": 0.06971484422683716, "sampling/importance_sampling_ratio/mean": 0.9949712753295898, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1192193403840065, "clip_ratio/low_mean": 0.08543961867690086, "clip_ratio/low_min": 0.08543961867690086, "clip_ratio/high_mean": 0.06054006889462471, "clip_ratio/high_max": 0.06054006889462471, "clip_ratio/region_mean": 0.14597968757152557, "reward_total_mean": 0.44581830501556396, "reward_meter_mean": 0.29853975772857666, "reward_meter_std": 0.3955845832824707, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9885627627372742, "reward_repeat_soft_std": 0.010870145633816719, "reward_judge_quality_mean": 0.6575000286102295, "reward_judge_quality_std": 0.2133910059928894, "reward_total_composite_mean": 0.44581830501556396, "reward_total_composite_std": 0.10586800426244736} {"timestamp_utc": "2026-04-13T09:22:10Z", "mode": "train", "global_step": 802, "epoch": 0.08056253139126067, "loss": 0.0174, "grad_norm": 10.958527565002441, "learning_rate": 7.572727272727274e-06, "num_tokens": 1418006.0, "completions/mean_length": 56.5, "completions/min_length": 45.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.5, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9016284942626953, "rewards/meter/std": 0.1466209888458252, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9821110963821411, "rewards/repeat_soft/std": 0.021698538213968277, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.22720351815223694, "rewards/total_composite/mean": 0.6615656614303589, "rewards/total_composite/std": 0.11730705201625824, "reward": 0.6615656614303589, "reward_std": 0.11730704456567764, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15427520871162415, "sampling/sampling_logp_difference/max": 2.4017622470855713, "sampling/importance_sampling_ratio/min": 0.09055822342634201, "sampling/importance_sampling_ratio/mean": 1.010503888130188, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8828255087137222, "clip_ratio/low_mean": 0.1184957604855299, "clip_ratio/low_min": 0.1184957604855299, "clip_ratio/high_mean": 0.036312622018158436, "clip_ratio/high_max": 0.036312622018158436, "clip_ratio/region_mean": 0.15480838250368834, "reward_total_mean": 0.6615656614303589, "reward_meter_mean": 0.9016284942626953, "reward_meter_std": 0.1466209888458252, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9821110963821411, "reward_repeat_soft_std": 0.021698538213968277, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.22720351815223694, "reward_total_composite_mean": 0.6615656614303589, "reward_total_composite_std": 0.11730705201625824} {"timestamp_utc": "2026-04-13T09:22:22Z", "mode": "train", "global_step": 803, "epoch": 0.08066298342541436, "loss": -0.1125, "grad_norm": 3.0111196041107178, "learning_rate": 7.56969696969697e-06, "num_tokens": 1419800.0, "completions/mean_length": 183.25, "completions/min_length": 53.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 73.66667175292969, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.4995984733104706, "rewards/meter/std": 0.3973415493965149, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.24800792336463928, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9811067581176758, "rewards/repeat_soft/std": 0.029663169756531715, "rewards/judge_quality/mean": 0.7024999856948853, "rewards/judge_quality/std": 0.4027317464351654, "rewards/total_composite/mean": 0.4504998028278351, "rewards/total_composite/std": 0.3490696847438812, "reward": 0.4504998028278351, "reward_std": 0.3490696847438812, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13643351197242737, "sampling/sampling_logp_difference/max": 1.9621081352233887, "sampling/importance_sampling_ratio/min": 0.1405617892742157, "sampling/importance_sampling_ratio/mean": 0.9995192289352417, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5766572505235672, "clip_ratio/low_mean": 0.029759816825389862, "clip_ratio/low_min": 0.029759816825389862, "clip_ratio/high_mean": 0.06271313689649105, "clip_ratio/high_max": 0.06271313689649105, "clip_ratio/region_mean": 0.09247295372188091, "reward_total_mean": 0.4504998028278351, "reward_meter_mean": 0.4995984733104706, "reward_meter_std": 0.3973415493965149, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.24800792336463928, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9811067581176758, "reward_repeat_soft_std": 0.029663169756531715, "reward_judge_quality_mean": 0.7024999856948853, "reward_judge_quality_std": 0.4027317464351654, "reward_total_composite_mean": 0.4504998028278351, "reward_total_composite_std": 0.3490696847438812} {"timestamp_utc": "2026-04-13T09:22:29Z", "mode": "train", "global_step": 804, "epoch": 0.08076343545956806, "loss": 0.0414, "grad_norm": 11.315879821777344, "learning_rate": 7.566666666666667e-06, "num_tokens": 1422235.0, "completions/mean_length": 112.375, "completions/min_length": 103.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.375, "completions/min_terminated_length": 103.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.7090272903442383, "rewards/meter/std": 0.2736465334892273, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9467700123786926, "rewards/repeat_soft/std": 0.030022691935300827, "rewards/judge_quality/mean": 0.6449999809265137, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.5997272729873657, "rewards/total_composite/std": 0.13130873441696167, "reward": 0.5997272729873657, "reward_std": 0.13130873441696167, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15653181076049805, "sampling/sampling_logp_difference/max": 2.0967979431152344, "sampling/importance_sampling_ratio/min": 0.12284916639328003, "sampling/importance_sampling_ratio/mean": 1.0141373872756958, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9335663095116615, "clip_ratio/low_mean": 0.06188603490591049, "clip_ratio/low_min": 0.06188603490591049, "clip_ratio/high_mean": 0.07365716155618429, "clip_ratio/high_max": 0.07365716155618429, "clip_ratio/region_mean": 0.13554319646209478, "reward_total_mean": 0.5997272729873657, "reward_meter_mean": 0.7090272903442383, "reward_meter_std": 0.2736465334892273, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9467700123786926, "reward_repeat_soft_std": 0.030022691935300827, "reward_judge_quality_mean": 0.6449999809265137, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.5997272729873657, "reward_total_composite_std": 0.13130873441696167} {"timestamp_utc": "2026-04-13T09:22:36Z", "mode": "train", "global_step": 805, "epoch": 0.08086388749372175, "loss": -0.0006, "grad_norm": 15.6214599609375, "learning_rate": 7.563636363636364e-06, "num_tokens": 1423947.0, "completions/mean_length": 45.0, "completions/min_length": 34.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.0, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.5403231978416443, "rewards/meter/std": 0.3556986153125763, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9451793432235718, "rewards/repeat_soft/std": 0.09756789356470108, "rewards/judge_quality/mean": 0.59375, "rewards/judge_quality/std": 0.22398583590984344, "rewards/total_composite/mean": 0.5799914598464966, "rewards/total_composite/std": 0.21093840897083282, "reward": 0.5799914598464966, "reward_std": 0.21093840897083282, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1565788835287094, "sampling/sampling_logp_difference/max": 1.6490745544433594, "sampling/importance_sampling_ratio/min": 0.19222772121429443, "sampling/importance_sampling_ratio/mean": 1.0035032033920288, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9338272884488106, "clip_ratio/low_mean": 0.08352437056601048, "clip_ratio/low_min": 0.08352437056601048, "clip_ratio/high_mean": 0.07671568915247917, "clip_ratio/high_max": 0.07671568915247917, "clip_ratio/region_mean": 0.16024005971848965, "reward_total_mean": 0.5799914598464966, "reward_meter_mean": 0.5403231978416443, "reward_meter_std": 0.3556986153125763, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9451793432235718, "reward_repeat_soft_std": 0.09756789356470108, "reward_judge_quality_mean": 0.59375, "reward_judge_quality_std": 0.22398583590984344, "reward_total_composite_mean": 0.5799914598464966, "reward_total_composite_std": 0.21093840897083282} {"timestamp_utc": "2026-04-13T09:22:42Z", "mode": "train", "global_step": 806, "epoch": 0.08096433952787543, "loss": -0.0222, "grad_norm": 13.966983795166016, "learning_rate": 7.560606060606062e-06, "num_tokens": 1425584.0, "completions/mean_length": 43.625, "completions/min_length": 41.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.625, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.7136768102645874, "rewards/meter/std": 0.3417411148548126, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9802080392837524, "rewards/repeat_soft/std": 0.03066585585474968, "rewards/judge_quality/mean": 0.6399999856948853, "rewards/judge_quality/std": 0.12224100530147552, "rewards/total_composite/mean": 0.6391867399215698, "rewards/total_composite/std": 0.15036608278751373, "reward": 0.6391867399215698, "reward_std": 0.15036608278751373, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16272421181201935, "sampling/sampling_logp_difference/max": 2.713101387023926, "sampling/importance_sampling_ratio/min": 0.20161476731300354, "sampling/importance_sampling_ratio/mean": 1.0206215381622314, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.962073840200901, "clip_ratio/low_mean": 0.08706512209028006, "clip_ratio/low_min": 0.08706512209028006, "clip_ratio/high_mean": 0.06618053466081619, "clip_ratio/high_max": 0.06618053466081619, "clip_ratio/region_mean": 0.15324565675109625, "reward_total_mean": 0.6391867399215698, "reward_meter_mean": 0.7136768102645874, "reward_meter_std": 0.3417411148548126, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9802080392837524, "reward_repeat_soft_std": 0.03066585585474968, "reward_judge_quality_mean": 0.6399999856948853, "reward_judge_quality_std": 0.12224100530147552, "reward_total_composite_mean": 0.6391867399215698, "reward_total_composite_std": 0.15036608278751373} {"timestamp_utc": "2026-04-13T09:22:54Z", "mode": "train", "global_step": 807, "epoch": 0.08106479156202913, "loss": -0.0503, "grad_norm": 4.909672737121582, "learning_rate": 7.557575757575758e-06, "num_tokens": 1427142.0, "completions/mean_length": 103.75, "completions/min_length": 39.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 45.42857360839844, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.7983464002609253, "rewards/meter/std": 0.28263235092163086, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.985405683517456, "rewards/repeat_soft/std": 0.010004878975450993, "rewards/judge_quality/mean": 0.5099999904632568, "rewards/judge_quality/std": 0.2865559160709381, "rewards/total_composite/mean": 0.5196365118026733, "rewards/total_composite/std": 0.36020785570144653, "reward": 0.5196365118026733, "reward_std": 0.36020782589912415, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14581044018268585, "sampling/sampling_logp_difference/max": 1.498145580291748, "sampling/importance_sampling_ratio/min": 0.22354432940483093, "sampling/importance_sampling_ratio/mean": 1.0206215381622314, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.772854670882225, "clip_ratio/low_mean": 0.02757973689585924, "clip_ratio/low_min": 0.02757973689585924, "clip_ratio/high_mean": 0.09412868414074183, "clip_ratio/high_max": 0.09412868414074183, "clip_ratio/region_mean": 0.12170842103660107, "reward_total_mean": 0.5196365118026733, "reward_meter_mean": 0.7983464002609253, "reward_meter_std": 0.28263235092163086, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.985405683517456, "reward_repeat_soft_std": 0.010004878975450993, "reward_judge_quality_mean": 0.5099999904632568, "reward_judge_quality_std": 0.2865559160709381, "reward_total_composite_mean": 0.5196365118026733, "reward_total_composite_std": 0.36020785570144653} {"timestamp_utc": "2026-04-13T09:23:01Z", "mode": "train", "global_step": 808, "epoch": 0.08116524359618282, "loss": -0.0245, "grad_norm": 13.79069995880127, "learning_rate": 7.5545454545454555e-06, "num_tokens": 1428709.0, "completions/mean_length": 49.875, "completions/min_length": 45.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.875, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.8329415917396545, "rewards/meter/std": 0.34195801615715027, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9917269945144653, "rewards/repeat_soft/std": 0.009106655605137348, "rewards/judge_quality/mean": 0.5637500286102295, "rewards/judge_quality/std": 0.22012579441070557, "rewards/total_composite/mean": 0.665428876876831, "rewards/total_composite/std": 0.19449499249458313, "reward": 0.665428876876831, "reward_std": 0.19449497759342194, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15418878197669983, "sampling/sampling_logp_difference/max": 5.072943687438965, "sampling/importance_sampling_ratio/min": 0.006263953633606434, "sampling/importance_sampling_ratio/mean": 1.0029226541519165, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9787449389696121, "clip_ratio/low_mean": 0.0927612129598856, "clip_ratio/low_min": 0.0927612129598856, "clip_ratio/high_mean": 0.03973354212939739, "clip_ratio/high_max": 0.03973354212939739, "clip_ratio/region_mean": 0.132494755089283, "reward_total_mean": 0.665428876876831, "reward_meter_mean": 0.8329415917396545, "reward_meter_std": 0.34195801615715027, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9917269945144653, "reward_repeat_soft_std": 0.009106655605137348, "reward_judge_quality_mean": 0.5637500286102295, "reward_judge_quality_std": 0.22012579441070557, "reward_total_composite_mean": 0.665428876876831, "reward_total_composite_std": 0.19449499249458313} {"timestamp_utc": "2026-04-13T09:23:08Z", "mode": "train", "global_step": 809, "epoch": 0.08126569563033652, "loss": -0.006, "grad_norm": 15.120176315307617, "learning_rate": 7.551515151515152e-06, "num_tokens": 1430435.0, "completions/mean_length": 53.75, "completions/min_length": 47.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.75, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.7933717966079712, "rewards/meter/std": 0.2962089776992798, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.91013103723526, "rewards/repeat_soft/std": 0.10067904740571976, "rewards/judge_quality/mean": 0.6200000047683716, "rewards/judge_quality/std": 0.22677870094776154, "rewards/total_composite/mean": 0.655008852481842, "rewards/total_composite/std": 0.1684381663799286, "reward": 0.655008852481842, "reward_std": 0.1684381514787674, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15119916200637817, "sampling/sampling_logp_difference/max": 1.5658235549926758, "sampling/importance_sampling_ratio/min": 0.20891588926315308, "sampling/importance_sampling_ratio/mean": 1.0017062425613403, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.051535002887249, "clip_ratio/low_mean": 0.06343050673604012, "clip_ratio/low_min": 0.06343050673604012, "clip_ratio/high_mean": 0.049440870992839336, "clip_ratio/high_max": 0.049440870992839336, "clip_ratio/region_mean": 0.11287137772887945, "reward_total_mean": 0.655008852481842, "reward_meter_mean": 0.7933717966079712, "reward_meter_std": 0.2962089776992798, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.91013103723526, "reward_repeat_soft_std": 0.10067904740571976, "reward_judge_quality_mean": 0.6200000047683716, "reward_judge_quality_std": 0.22677870094776154, "reward_total_composite_mean": 0.655008852481842, "reward_total_composite_std": 0.1684381663799286} {"timestamp_utc": "2026-04-13T09:23:16Z", "mode": "train", "global_step": 810, "epoch": 0.08136614766449021, "loss": -0.0396, "grad_norm": 25.197595596313477, "learning_rate": 7.548484848484849e-06, "num_tokens": 1431944.0, "completions/mean_length": 26.625, "completions/min_length": 20.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.625, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.7921427488327026, "rewards/meter/std": 0.35519370436668396, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9307898283004761, "rewards/repeat_soft/std": 0.08015943318605423, "rewards/judge_quality/mean": 0.4424999952316284, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.5676900744438171, "rewards/total_composite/std": 0.10124175250530243, "reward": 0.5676900744438171, "reward_std": 0.10124175250530243, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1733849048614502, "sampling/sampling_logp_difference/max": 1.4316306114196777, "sampling/importance_sampling_ratio/min": 0.23891901969909668, "sampling/importance_sampling_ratio/mean": 0.9951972961425781, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1499949097633362, "clip_ratio/low_mean": 0.046875, "clip_ratio/low_min": 0.046875, "clip_ratio/high_mean": 0.1721633835695684, "clip_ratio/high_max": 0.1721633835695684, "clip_ratio/region_mean": 0.2190383835695684, "reward_total_mean": 0.5676900744438171, "reward_meter_mean": 0.7921427488327026, "reward_meter_std": 0.35519370436668396, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9307898283004761, "reward_repeat_soft_std": 0.08015943318605423, "reward_judge_quality_mean": 0.4424999952316284, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.5676900744438171, "reward_total_composite_std": 0.10124175250530243} {"timestamp_utc": "2026-04-13T09:23:22Z", "mode": "train", "global_step": 811, "epoch": 0.0814665996986439, "loss": 0.0569, "grad_norm": 14.960481643676758, "learning_rate": 7.545454545454546e-06, "num_tokens": 1433724.0, "completions/mean_length": 47.5, "completions/min_length": 41.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.5, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.3600141406059265, "rewards/meter/std": 0.3861323893070221, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9460911750793457, "rewards/repeat_soft/std": 0.06058942526578903, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.4455781877040863, "rewards/total_composite/std": 0.11271742731332779, "reward": 0.4455781877040863, "reward_std": 0.11271742731332779, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14823615550994873, "sampling/sampling_logp_difference/max": 1.5308938026428223, "sampling/importance_sampling_ratio/min": 0.22016087174415588, "sampling/importance_sampling_ratio/mean": 1.015677571296692, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8647165969014168, "clip_ratio/low_mean": 0.10145791340619326, "clip_ratio/low_min": 0.10145791340619326, "clip_ratio/high_mean": 0.0511738546192646, "clip_ratio/high_max": 0.0511738546192646, "clip_ratio/region_mean": 0.15263176802545786, "reward_total_mean": 0.4455781877040863, "reward_meter_mean": 0.3600141406059265, "reward_meter_std": 0.3861323893070221, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9460911750793457, "reward_repeat_soft_std": 0.06058942526578903, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.4455781877040863, "reward_total_composite_std": 0.11271742731332779} {"timestamp_utc": "2026-04-13T09:23:34Z", "mode": "train", "global_step": 812, "epoch": 0.08156705173279759, "loss": -0.0049, "grad_norm": 5.505806922912598, "learning_rate": 7.542424242424244e-06, "num_tokens": 1435525.0, "completions/mean_length": 108.125, "completions/min_length": 40.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 50.42857360839844, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.15675875544548035, "rewards/meter/std": 0.2787606120109558, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9964594841003418, "rewards/repeat_soft/std": 0.004860998596996069, "rewards/judge_quality/mean": 0.4987499713897705, "rewards/judge_quality/std": 0.28965190052986145, "rewards/total_composite/mean": 0.39452141523361206, "rewards/total_composite/std": 0.07557890564203262, "reward": 0.39452141523361206, "reward_std": 0.07557890564203262, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1547120362520218, "sampling/sampling_logp_difference/max": 1.3801994323730469, "sampling/importance_sampling_ratio/min": 0.25152841210365295, "sampling/importance_sampling_ratio/mean": 1.0177757740020752, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7508349493145943, "clip_ratio/low_mean": 0.08179171429947019, "clip_ratio/low_min": 0.08179171429947019, "clip_ratio/high_mean": 0.051739130169153214, "clip_ratio/high_max": 0.051739130169153214, "clip_ratio/region_mean": 0.1335308444686234, "reward_total_mean": 0.39452141523361206, "reward_meter_mean": 0.15675875544548035, "reward_meter_std": 0.2787606120109558, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9964594841003418, "reward_repeat_soft_std": 0.004860998596996069, "reward_judge_quality_mean": 0.4987499713897705, "reward_judge_quality_std": 0.28965190052986145, "reward_total_composite_mean": 0.39452141523361206, "reward_total_composite_std": 0.07557890564203262} {"timestamp_utc": "2026-04-13T09:23:46Z", "mode": "train", "global_step": 813, "epoch": 0.08166750376695128, "loss": -0.1233, "grad_norm": 4.601381778717041, "learning_rate": 7.53939393939394e-06, "num_tokens": 1437221.0, "completions/mean_length": 117.0, "completions/min_length": 53.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 60.57143020629883, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.6576850414276123, "rewards/meter/std": 0.28868722915649414, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.2357022762298584, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9237185716629028, "rewards/repeat_soft/std": 0.05879615247249603, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.23439893126487732, "rewards/total_composite/mean": 0.5077955722808838, "rewards/total_composite/std": 0.2563643157482147, "reward": 0.5077955722808838, "reward_std": 0.2563643157482147, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1729813665151596, "sampling/sampling_logp_difference/max": 3.193268060684204, "sampling/importance_sampling_ratio/min": 0.04103753715753555, "sampling/importance_sampling_ratio/mean": 0.9843983054161072, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6876777485013008, "clip_ratio/low_mean": 0.04267744906246662, "clip_ratio/low_min": 0.04267744906246662, "clip_ratio/high_mean": 0.10265432484447956, "clip_ratio/high_max": 0.10265432484447956, "clip_ratio/region_mean": 0.14533177390694618, "reward_total_mean": 0.5077955722808838, "reward_meter_mean": 0.6576850414276123, "reward_meter_std": 0.28868722915649414, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.2357022762298584, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9237185716629028, "reward_repeat_soft_std": 0.05879615247249603, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.23439893126487732, "reward_total_composite_mean": 0.5077955722808838, "reward_total_composite_std": 0.2563643157482147} {"timestamp_utc": "2026-04-13T09:23:57Z", "mode": "train", "global_step": 814, "epoch": 0.08176795580110498, "loss": -0.13, "grad_norm": 3.343351364135742, "learning_rate": 7.536363636363637e-06, "num_tokens": 1438797.0, "completions/mean_length": 111.0, "completions/min_length": 41.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 53.71428680419922, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.6222841739654541, "rewards/meter/std": 0.3936542272567749, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9862548112869263, "rewards/repeat_soft/std": 0.014779746532440186, "rewards/judge_quality/mean": 0.41874998807907104, "rewards/judge_quality/std": 0.18074746429920197, "rewards/total_composite/mean": 0.4458679258823395, "rewards/total_composite/std": 0.20553505420684814, "reward": 0.4458679258823395, "reward_std": 0.20553503930568695, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18630266189575195, "sampling/sampling_logp_difference/max": 1.6068196296691895, "sampling/importance_sampling_ratio/min": 0.20052434504032135, "sampling/importance_sampling_ratio/mean": 1.0044987201690674, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1804257184267044, "clip_ratio/low_mean": 0.061155762523412704, "clip_ratio/low_min": 0.061155762523412704, "clip_ratio/high_mean": 0.1013546446338296, "clip_ratio/high_max": 0.1013546446338296, "clip_ratio/region_mean": 0.1625104071572423, "reward_total_mean": 0.4458679258823395, "reward_meter_mean": 0.6222841739654541, "reward_meter_std": 0.3936542272567749, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9862548112869263, "reward_repeat_soft_std": 0.014779746532440186, "reward_judge_quality_mean": 0.41874998807907104, "reward_judge_quality_std": 0.18074746429920197, "reward_total_composite_mean": 0.4458679258823395, "reward_total_composite_std": 0.20553505420684814} {"timestamp_utc": "2026-04-13T09:24:03Z", "mode": "train", "global_step": 815, "epoch": 0.08186840783525866, "loss": 0.0498, "grad_norm": 19.04191780090332, "learning_rate": 7.533333333333334e-06, "num_tokens": 1440461.0, "completions/mean_length": 25.0, "completions/min_length": 20.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.0, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.7894392013549805, "rewards/meter/std": 0.3342284560203552, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8165859580039978, "rewards/repeat_soft/std": 0.12885448336601257, "rewards/judge_quality/mean": 0.8025000095367432, "rewards/judge_quality/std": 0.21756774187088013, "rewards/total_composite/mean": 0.7297252416610718, "rewards/total_composite/std": 0.2254355251789093, "reward": 0.7297252416610718, "reward_std": 0.22543549537658691, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11707991361618042, "sampling/sampling_logp_difference/max": 1.4313504695892334, "sampling/importance_sampling_ratio/min": 0.23898597061634064, "sampling/importance_sampling_ratio/mean": 0.9947558641433716, "sampling/importance_sampling_ratio/max": 1.6051779985427856, "entropy": 0.6842283755540848, "clip_ratio/low_mean": 0.033491379115730524, "clip_ratio/low_min": 0.033491379115730524, "clip_ratio/high_mean": 0.08551282249391079, "clip_ratio/high_max": 0.08551282249391079, "clip_ratio/region_mean": 0.11900420160964131, "reward_total_mean": 0.7297252416610718, "reward_meter_mean": 0.7894392013549805, "reward_meter_std": 0.3342284560203552, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8165859580039978, "reward_repeat_soft_std": 0.12885448336601257, "reward_judge_quality_mean": 0.8025000095367432, "reward_judge_quality_std": 0.21756774187088013, "reward_total_composite_mean": 0.7297252416610718, "reward_total_composite_std": 0.2254355251789093} {"timestamp_utc": "2026-04-13T09:24:09Z", "mode": "train", "global_step": 816, "epoch": 0.08196885986941235, "loss": 0.0396, "grad_norm": 14.968425750732422, "learning_rate": 7.530303030303031e-06, "num_tokens": 1442114.0, "completions/mean_length": 50.625, "completions/min_length": 33.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.625, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.49092987179756165, "rewards/meter/std": 0.28902894258499146, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9823921918869019, "rewards/repeat_soft/std": 0.028371434658765793, "rewards/judge_quality/mean": 0.5087499618530273, "rewards/judge_quality/std": 0.16617010533809662, "rewards/total_composite/mean": 0.5222827792167664, "rewards/total_composite/std": 0.15004116296768188, "reward": 0.5222827792167664, "reward_std": 0.15004117786884308, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20552361011505127, "sampling/sampling_logp_difference/max": 1.8502120971679688, "sampling/importance_sampling_ratio/min": 0.1572038233280182, "sampling/importance_sampling_ratio/mean": 1.0396384000778198, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.494948334991932, "clip_ratio/low_mean": 0.10349871683865786, "clip_ratio/low_min": 0.10349871683865786, "clip_ratio/high_mean": 0.0664410050958395, "clip_ratio/high_max": 0.0664410050958395, "clip_ratio/region_mean": 0.16993972193449736, "reward_total_mean": 0.5222827792167664, "reward_meter_mean": 0.49092987179756165, "reward_meter_std": 0.28902894258499146, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9823921918869019, "reward_repeat_soft_std": 0.028371434658765793, "reward_judge_quality_mean": 0.5087499618530273, "reward_judge_quality_std": 0.16617010533809662, "reward_total_composite_mean": 0.5222827792167664, "reward_total_composite_std": 0.15004116296768188} {"timestamp_utc": "2026-04-13T09:24:16Z", "mode": "train", "global_step": 817, "epoch": 0.08206931190356605, "loss": 0.0807, "grad_norm": 12.185218811035156, "learning_rate": 7.5272727272727274e-06, "num_tokens": 1444393.0, "completions/mean_length": 95.875, "completions/min_length": 76.0, "completions/max_length": 118.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.875, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 118.0, "rewards/meter/mean": 0.7116522789001465, "rewards/meter/std": 0.2779863178730011, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.13363061845302582, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9482159614562988, "rewards/repeat_soft/std": 0.034769054502248764, "rewards/judge_quality/mean": 0.6450000405311584, "rewards/judge_quality/std": 0.19820626080036163, "rewards/total_composite/mean": 0.6138795614242554, "rewards/total_composite/std": 0.1565939038991928, "reward": 0.6138795614242554, "reward_std": 0.1565939038991928, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1762741208076477, "sampling/sampling_logp_difference/max": 2.9817864894866943, "sampling/importance_sampling_ratio/min": 0.05070217698812485, "sampling/importance_sampling_ratio/mean": 1.0143365859985352, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0195010900497437, "clip_ratio/low_mean": 0.103886135853827, "clip_ratio/low_min": 0.103886135853827, "clip_ratio/high_mean": 0.049659901298582554, "clip_ratio/high_max": 0.049659901298582554, "clip_ratio/region_mean": 0.15354603715240955, "reward_total_mean": 0.6138795614242554, "reward_meter_mean": 0.7116522789001465, "reward_meter_std": 0.2779863178730011, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.13363061845302582, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9482159614562988, "reward_repeat_soft_std": 0.034769054502248764, "reward_judge_quality_mean": 0.6450000405311584, "reward_judge_quality_std": 0.19820626080036163, "reward_total_composite_mean": 0.6138795614242554, "reward_total_composite_std": 0.1565939038991928} {"timestamp_utc": "2026-04-13T09:24:22Z", "mode": "train", "global_step": 818, "epoch": 0.08216976393771974, "loss": 0.1043, "grad_norm": 16.24044418334961, "learning_rate": 7.524242424242425e-06, "num_tokens": 1446016.0, "completions/mean_length": 44.875, "completions/min_length": 40.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.875, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.42738354206085205, "rewards/meter/std": 0.2925856113433838, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9781241416931152, "rewards/repeat_soft/std": 0.01897178590297699, "rewards/judge_quality/mean": 0.8025000095367432, "rewards/judge_quality/std": 0.21756774187088013, "rewards/total_composite/mean": 0.5775432586669922, "rewards/total_composite/std": 0.1707666516304016, "reward": 0.5775432586669922, "reward_std": 0.17076663672924042, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20095308125019073, "sampling/sampling_logp_difference/max": 2.370021343231201, "sampling/importance_sampling_ratio/min": 0.0934787318110466, "sampling/importance_sampling_ratio/mean": 1.0148979425430298, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1817000210285187, "clip_ratio/low_mean": 0.13013580441474915, "clip_ratio/low_min": 0.13013580441474915, "clip_ratio/high_mean": 0.07315891608595848, "clip_ratio/high_max": 0.07315891608595848, "clip_ratio/region_mean": 0.20329472050070763, "reward_total_mean": 0.5775432586669922, "reward_meter_mean": 0.42738354206085205, "reward_meter_std": 0.2925856113433838, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9781241416931152, "reward_repeat_soft_std": 0.01897178590297699, "reward_judge_quality_mean": 0.8025000095367432, "reward_judge_quality_std": 0.21756774187088013, "reward_total_composite_mean": 0.5775432586669922, "reward_total_composite_std": 0.1707666516304016} {"timestamp_utc": "2026-04-13T09:24:33Z", "mode": "train", "global_step": 819, "epoch": 0.08227021597187344, "loss": -0.1303, "grad_norm": 3.4475998878479004, "learning_rate": 7.521212121212121e-06, "num_tokens": 1447870.0, "completions/mean_length": 181.75, "completions/min_length": 65.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 71.66667175292969, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.5192000865936279, "rewards/meter/std": 0.3829488754272461, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9459807872772217, "rewards/repeat_soft/std": 0.02395819127559662, "rewards/judge_quality/mean": 0.45250001549720764, "rewards/judge_quality/std": 0.33065953850746155, "rewards/total_composite/mean": 0.4163227677345276, "rewards/total_composite/std": 0.2916165888309479, "reward": 0.4163227677345276, "reward_std": 0.2916165888309479, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1499626338481903, "sampling/sampling_logp_difference/max": 2.4193878173828125, "sampling/importance_sampling_ratio/min": 0.08897607028484344, "sampling/importance_sampling_ratio/mean": 1.01922607421875, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6223423704504967, "clip_ratio/low_mean": 0.03234265837818384, "clip_ratio/low_min": 0.03234265837818384, "clip_ratio/high_mean": 0.06764488946646452, "clip_ratio/high_max": 0.06764488946646452, "clip_ratio/region_mean": 0.09998754784464836, "reward_total_mean": 0.4163227677345276, "reward_meter_mean": 0.5192000865936279, "reward_meter_std": 0.3829488754272461, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9459807872772217, "reward_repeat_soft_std": 0.02395819127559662, "reward_judge_quality_mean": 0.45250001549720764, "reward_judge_quality_std": 0.33065953850746155, "reward_total_composite_mean": 0.4163227677345276, "reward_total_composite_std": 0.2916165888309479} {"timestamp_utc": "2026-04-13T09:24:44Z", "mode": "train", "global_step": 820, "epoch": 0.08237066800602712, "loss": -0.1359, "grad_norm": 3.6564669609069824, "learning_rate": 7.518181818181819e-06, "num_tokens": 1450045.0, "completions/mean_length": 149.875, "completions/min_length": 71.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 98.14286041259766, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 138.0, "rewards/meter/mean": 0.6863880157470703, "rewards/meter/std": 0.37522250413894653, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.13363061845302582, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8512330651283264, "rewards/repeat_soft/std": 0.11977135390043259, "rewards/judge_quality/mean": 0.4737499952316284, "rewards/judge_quality/std": 0.254611074924469, "rewards/total_composite/mean": 0.5125121474266052, "rewards/total_composite/std": 0.2600172162055969, "reward": 0.5125121474266052, "reward_std": 0.2600172162055969, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14205388724803925, "sampling/sampling_logp_difference/max": 3.338271141052246, "sampling/importance_sampling_ratio/min": 0.035498276352882385, "sampling/importance_sampling_ratio/mean": 0.9941490888595581, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7078659161925316, "clip_ratio/low_mean": 0.03297199681401253, "clip_ratio/low_min": 0.03297199681401253, "clip_ratio/high_mean": 0.09131707530468702, "clip_ratio/high_max": 0.09131707530468702, "clip_ratio/region_mean": 0.12428907211869955, "reward_total_mean": 0.5125121474266052, "reward_meter_mean": 0.6863880157470703, "reward_meter_std": 0.37522250413894653, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.13363061845302582, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8512330651283264, "reward_repeat_soft_std": 0.11977135390043259, "reward_judge_quality_mean": 0.4737499952316284, "reward_judge_quality_std": 0.254611074924469, "reward_total_composite_mean": 0.5125121474266052, "reward_total_composite_std": 0.2600172162055969} {"timestamp_utc": "2026-04-13T09:24:51Z", "mode": "train", "global_step": 821, "epoch": 0.08247112004018081, "loss": -0.0201, "grad_norm": 9.674942970275879, "learning_rate": 7.515151515151516e-06, "num_tokens": 1452235.0, "completions/mean_length": 90.75, "completions/min_length": 83.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.75, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.4520382881164551, "rewards/meter/std": 0.19169962406158447, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9560103416442871, "rewards/repeat_soft/std": 0.05141996592283249, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.47455012798309326, "rewards/total_composite/std": 0.06853678822517395, "reward": 0.47455012798309326, "reward_std": 0.06853678077459335, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15331177413463593, "sampling/sampling_logp_difference/max": 1.9285740852355957, "sampling/importance_sampling_ratio/min": 0.14535531401634216, "sampling/importance_sampling_ratio/mean": 1.0141502618789673, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8684991300106049, "clip_ratio/low_mean": 0.09285540506243706, "clip_ratio/low_min": 0.09285540506243706, "clip_ratio/high_mean": 0.07472315896302462, "clip_ratio/high_max": 0.07472315896302462, "clip_ratio/region_mean": 0.16757856402546167, "reward_total_mean": 0.47455012798309326, "reward_meter_mean": 0.4520382881164551, "reward_meter_std": 0.19169962406158447, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9560103416442871, "reward_repeat_soft_std": 0.05141996592283249, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.47455012798309326, "reward_total_composite_std": 0.06853678822517395} {"timestamp_utc": "2026-04-13T09:24:58Z", "mode": "train", "global_step": 822, "epoch": 0.0825715720743345, "loss": 0.0216, "grad_norm": 19.24358558654785, "learning_rate": 7.512121212121213e-06, "num_tokens": 1453858.0, "completions/mean_length": 28.875, "completions/min_length": 23.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.875, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.7364631295204163, "rewards/meter/std": 0.40704989433288574, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9546753168106079, "rewards/repeat_soft/std": 0.013788825832307339, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.0645727664232254, "rewards/total_composite/mean": 0.5358984470367432, "rewards/total_composite/std": 0.11425299942493439, "reward": 0.5358984470367432, "reward_std": 0.11425299197435379, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20006316900253296, "sampling/sampling_logp_difference/max": 1.5987379550933838, "sampling/importance_sampling_ratio/min": 0.20215147733688354, "sampling/importance_sampling_ratio/mean": 1.0060007572174072, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3630374073982239, "clip_ratio/low_mean": 0.05968630267307162, "clip_ratio/low_min": 0.05968630267307162, "clip_ratio/high_mean": 0.15672273561358452, "clip_ratio/high_max": 0.15672273561358452, "clip_ratio/region_mean": 0.21640903828665614, "reward_total_mean": 0.5358984470367432, "reward_meter_mean": 0.7364631295204163, "reward_meter_std": 0.40704989433288574, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9546753168106079, "reward_repeat_soft_std": 0.013788825832307339, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.0645727664232254, "reward_total_composite_mean": 0.5358984470367432, "reward_total_composite_std": 0.11425299942493439} {"timestamp_utc": "2026-04-13T09:25:04Z", "mode": "train", "global_step": 823, "epoch": 0.0826720241084882, "loss": 0.0574, "grad_norm": 9.428529739379883, "learning_rate": 7.509090909090909e-06, "num_tokens": 1456342.0, "completions/mean_length": 98.5, "completions/min_length": 85.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.5, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.791350245475769, "rewards/meter/std": 0.2780003249645233, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9728424549102783, "rewards/repeat_soft/std": 0.011104121804237366, "rewards/judge_quality/mean": 0.7237499952316284, "rewards/judge_quality/std": 0.1418185532093048, "rewards/total_composite/mean": 0.6681636571884155, "rewards/total_composite/std": 0.12867885828018188, "reward": 0.6681636571884155, "reward_std": 0.12867885828018188, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12085013091564178, "sampling/sampling_logp_difference/max": 1.7185745239257812, "sampling/importance_sampling_ratio/min": 0.1881616860628128, "sampling/importance_sampling_ratio/mean": 1.0084378719329834, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5888826176524162, "clip_ratio/low_mean": 0.044817473739385605, "clip_ratio/low_min": 0.044817473739385605, "clip_ratio/high_mean": 0.06585449632257223, "clip_ratio/high_max": 0.06585449632257223, "clip_ratio/region_mean": 0.11067197006195784, "reward_total_mean": 0.6681636571884155, "reward_meter_mean": 0.791350245475769, "reward_meter_std": 0.2780003249645233, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9728424549102783, "reward_repeat_soft_std": 0.011104121804237366, "reward_judge_quality_mean": 0.7237499952316284, "reward_judge_quality_std": 0.1418185532093048, "reward_total_composite_mean": 0.6681636571884155, "reward_total_composite_std": 0.12867885828018188} {"timestamp_utc": "2026-04-13T09:25:11Z", "mode": "train", "global_step": 824, "epoch": 0.08277247614264188, "loss": 0.0782, "grad_norm": 18.671138763427734, "learning_rate": 7.5060606060606065e-06, "num_tokens": 1458085.0, "completions/mean_length": 37.875, "completions/min_length": 28.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.875, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.6317110061645508, "rewards/meter/std": 0.4083610773086548, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9670195579528809, "rewards/repeat_soft/std": 0.04772603139281273, "rewards/judge_quality/mean": 0.6775000095367432, "rewards/judge_quality/std": 0.25949129462242126, "rewards/total_composite/mean": 0.645653486251831, "rewards/total_composite/std": 0.2172938585281372, "reward": 0.645653486251831, "reward_std": 0.2172938585281372, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16183391213417053, "sampling/sampling_logp_difference/max": 1.4298913478851318, "sampling/importance_sampling_ratio/min": 0.23933492600917816, "sampling/importance_sampling_ratio/mean": 1.0175118446350098, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7881267480552197, "clip_ratio/low_mean": 0.05909091094508767, "clip_ratio/low_min": 0.05909091094508767, "clip_ratio/high_mean": 0.08884947653859854, "clip_ratio/high_max": 0.08884947653859854, "clip_ratio/region_mean": 0.1479403874836862, "reward_total_mean": 0.645653486251831, "reward_meter_mean": 0.6317110061645508, "reward_meter_std": 0.4083610773086548, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9670195579528809, "reward_repeat_soft_std": 0.04772603139281273, "reward_judge_quality_mean": 0.6775000095367432, "reward_judge_quality_std": 0.25949129462242126, "reward_total_composite_mean": 0.645653486251831, "reward_total_composite_std": 0.2172938585281372} {"timestamp_utc": "2026-04-13T09:25:23Z", "mode": "train", "global_step": 825, "epoch": 0.08287292817679558, "loss": -0.1015, "grad_norm": 3.5623085498809814, "learning_rate": 7.503030303030303e-06, "num_tokens": 1459880.0, "completions/mean_length": 107.375, "completions/min_length": 47.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 49.57143020629883, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.5495851635932922, "rewards/meter/std": 0.44300636649131775, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9228882193565369, "rewards/repeat_soft/std": 0.07558905333280563, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.22947145998477936, "rewards/total_composite/mean": 0.49460169672966003, "rewards/total_composite/std": 0.2527218461036682, "reward": 0.49460169672966003, "reward_std": 0.2527218461036682, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14410503208637238, "sampling/sampling_logp_difference/max": 1.8627885580062866, "sampling/importance_sampling_ratio/min": 0.15523913502693176, "sampling/importance_sampling_ratio/mean": 1.0030133724212646, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6821320280432701, "clip_ratio/low_mean": 0.03609848441556096, "clip_ratio/low_min": 0.03609848441556096, "clip_ratio/high_mean": 0.07457344699651003, "clip_ratio/high_max": 0.07457344699651003, "clip_ratio/region_mean": 0.11067193141207099, "reward_total_mean": 0.49460169672966003, "reward_meter_mean": 0.5495851635932922, "reward_meter_std": 0.44300636649131775, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9228882193565369, "reward_repeat_soft_std": 0.07558905333280563, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.22947145998477936, "reward_total_composite_mean": 0.49460169672966003, "reward_total_composite_std": 0.2527218461036682} {"timestamp_utc": "2026-04-13T09:25:34Z", "mode": "train", "global_step": 826, "epoch": 0.08297338021094927, "loss": -0.0448, "grad_norm": 126.64268493652344, "learning_rate": 7.500000000000001e-06, "num_tokens": 1461737.0, "completions/mean_length": 127.125, "completions/min_length": 50.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 72.14286041259766, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.8361838459968567, "rewards/meter/std": 0.20359478890895844, "rewards/count_adherence/mean": 0.7916666865348816, "rewards/count_adherence/std": 0.24800792336463928, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9886918663978577, "rewards/repeat_soft/std": 0.00916048139333725, "rewards/judge_quality/mean": 0.4687500298023224, "rewards/judge_quality/std": 0.19037088751792908, "rewards/total_composite/mean": 0.5030023455619812, "rewards/total_composite/std": 0.25660642981529236, "reward": 0.5030023455619812, "reward_std": 0.25660642981529236, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12969906628131866, "sampling/sampling_logp_difference/max": 1.141408920288086, "sampling/importance_sampling_ratio/min": 0.31936874985694885, "sampling/importance_sampling_ratio/mean": 1.0196263790130615, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7928005829453468, "clip_ratio/low_mean": 0.03452381119132042, "clip_ratio/low_min": 0.03452381119132042, "clip_ratio/high_mean": 0.08575281407684088, "clip_ratio/high_max": 0.08575281407684088, "clip_ratio/region_mean": 0.1202766252681613, "reward_total_mean": 0.5030023455619812, "reward_meter_mean": 0.8361838459968567, "reward_meter_std": 0.20359478890895844, "reward_count_adherence_mean": 0.7916666865348816, "reward_count_adherence_std": 0.24800792336463928, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9886918663978577, "reward_repeat_soft_std": 0.00916048139333725, "reward_judge_quality_mean": 0.4687500298023224, "reward_judge_quality_std": 0.19037088751792908, "reward_total_composite_mean": 0.5030023455619812, "reward_total_composite_std": 0.25660642981529236} {"timestamp_utc": "2026-04-13T09:25:46Z", "mode": "train", "global_step": 827, "epoch": 0.08307383224510297, "loss": -0.1233, "grad_norm": 3.053903818130493, "learning_rate": 7.496969696969698e-06, "num_tokens": 1463335.0, "completions/mean_length": 113.75, "completions/min_length": 50.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 56.857147216796875, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.7329822182655334, "rewards/meter/std": 0.3762177526950836, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8214617967605591, "rewards/repeat_soft/std": 0.09373597055673599, "rewards/judge_quality/mean": 0.3387500047683716, "rewards/judge_quality/std": 0.14327171444892883, "rewards/total_composite/mean": 0.426153302192688, "rewards/total_composite/std": 0.19938506186008453, "reward": 0.426153302192688, "reward_std": 0.19938506186008453, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16028787195682526, "sampling/sampling_logp_difference/max": 1.5074095726013184, "sampling/importance_sampling_ratio/min": 0.22148297727108002, "sampling/importance_sampling_ratio/mean": 1.0096970796585083, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1295492947101593, "clip_ratio/low_mean": 0.021753935143351555, "clip_ratio/low_min": 0.021753935143351555, "clip_ratio/high_mean": 0.0876676170155406, "clip_ratio/high_max": 0.0876676170155406, "clip_ratio/region_mean": 0.10942155215889215, "reward_total_mean": 0.426153302192688, "reward_meter_mean": 0.7329822182655334, "reward_meter_std": 0.3762177526950836, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8214617967605591, "reward_repeat_soft_std": 0.09373597055673599, "reward_judge_quality_mean": 0.3387500047683716, "reward_judge_quality_std": 0.14327171444892883, "reward_total_composite_mean": 0.426153302192688, "reward_total_composite_std": 0.19938506186008453} {"timestamp_utc": "2026-04-13T09:25:53Z", "mode": "train", "global_step": 828, "epoch": 0.08317428427925666, "loss": 0.096, "grad_norm": 12.640822410583496, "learning_rate": 7.493939393939395e-06, "num_tokens": 1465031.0, "completions/mean_length": 54.0, "completions/min_length": 37.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.5518558025360107, "rewards/meter/std": 0.38216036558151245, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9843226671218872, "rewards/repeat_soft/std": 0.0100281722843647, "rewards/judge_quality/mean": 0.7325000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.6428855657577515, "rewards/total_composite/std": 0.24778501689434052, "reward": 0.6428855657577515, "reward_std": 0.24778501689434052, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1674165427684784, "sampling/sampling_logp_difference/max": 2.5901987552642822, "sampling/importance_sampling_ratio/min": 0.07500512897968292, "sampling/importance_sampling_ratio/mean": 1.0255234241485596, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2542723193764687, "clip_ratio/low_mean": 0.09152952954173088, "clip_ratio/low_min": 0.09152952954173088, "clip_ratio/high_mean": 0.07294967584311962, "clip_ratio/high_max": 0.07294967584311962, "clip_ratio/region_mean": 0.1644792053848505, "reward_total_mean": 0.6428855657577515, "reward_meter_mean": 0.5518558025360107, "reward_meter_std": 0.38216036558151245, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9843226671218872, "reward_repeat_soft_std": 0.0100281722843647, "reward_judge_quality_mean": 0.7325000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.6428855657577515, "reward_total_composite_std": 0.24778501689434052} {"timestamp_utc": "2026-04-13T09:26:04Z", "mode": "train", "global_step": 829, "epoch": 0.08327473631341034, "loss": -0.0929, "grad_norm": 3.071845054626465, "learning_rate": 7.490909090909092e-06, "num_tokens": 1466437.0, "completions/mean_length": 91.75, "completions/min_length": 30.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 31.71428680419922, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.8275455236434937, "rewards/meter/std": 0.33318546414375305, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9912638664245605, "rewards/repeat_soft/std": 0.01396881602704525, "rewards/judge_quality/mean": 0.5437500476837158, "rewards/judge_quality/std": 0.29861289262771606, "rewards/total_composite/mean": 0.6351829767227173, "rewards/total_composite/std": 0.2909258306026459, "reward": 0.6351829767227173, "reward_std": 0.2909258008003235, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1065325066447258, "sampling/sampling_logp_difference/max": 1.6565746068954468, "sampling/importance_sampling_ratio/min": 0.19079139828681946, "sampling/importance_sampling_ratio/mean": 1.0092597007751465, "sampling/importance_sampling_ratio/max": 1.9726223945617676, "entropy": 0.39859044924378395, "clip_ratio/low_mean": 0.04791666753590107, "clip_ratio/low_min": 0.04791666753590107, "clip_ratio/high_mean": 0.034446023404598236, "clip_ratio/high_max": 0.034446023404598236, "clip_ratio/region_mean": 0.0823626909404993, "reward_total_mean": 0.6351829767227173, "reward_meter_mean": 0.8275455236434937, "reward_meter_std": 0.33318546414375305, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9912638664245605, "reward_repeat_soft_std": 0.01396881602704525, "reward_judge_quality_mean": 0.5437500476837158, "reward_judge_quality_std": 0.29861289262771606, "reward_total_composite_mean": 0.6351829767227173, "reward_total_composite_std": 0.2909258306026459} {"timestamp_utc": "2026-04-13T09:26:12Z", "mode": "train", "global_step": 830, "epoch": 0.08337518834756404, "loss": 0.063, "grad_norm": 9.040604591369629, "learning_rate": 7.487878787878788e-06, "num_tokens": 1468904.0, "completions/mean_length": 107.375, "completions/min_length": 102.0, "completions/max_length": 125.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.375, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.9646186232566833, "rewards/meter/std": 0.043388731777668, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8325355052947998, "rewards/repeat_soft/std": 0.14555595815181732, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.548221230506897, "rewards/total_composite/std": 0.020647374913096428, "reward": 0.548221230506897, "reward_std": 0.020647387951612473, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13437898457050323, "sampling/sampling_logp_difference/max": 3.2449071407318115, "sampling/importance_sampling_ratio/min": 0.03897218033671379, "sampling/importance_sampling_ratio/mean": 0.9994519352912903, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7556704059243202, "clip_ratio/low_mean": 0.04775570100173354, "clip_ratio/low_min": 0.04775570100173354, "clip_ratio/high_mean": 0.08244301937520504, "clip_ratio/high_max": 0.08244301937520504, "clip_ratio/region_mean": 0.13019872037693858, "reward_total_mean": 0.548221230506897, "reward_meter_mean": 0.9646186232566833, "reward_meter_std": 0.043388731777668, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8325355052947998, "reward_repeat_soft_std": 0.14555595815181732, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.548221230506897, "reward_total_composite_std": 0.020647374913096428} {"timestamp_utc": "2026-04-13T09:26:19Z", "mode": "train", "global_step": 831, "epoch": 0.08347564038171773, "loss": 0.0236, "grad_norm": 10.653721809387207, "learning_rate": 7.484848484848486e-06, "num_tokens": 1470919.0, "completions/mean_length": 71.875, "completions/min_length": 54.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.875, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.5295940637588501, "rewards/meter/std": 0.3171357214450836, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9773417115211487, "rewards/repeat_soft/std": 0.015895428135991096, "rewards/judge_quality/mean": 0.6737499833106995, "rewards/judge_quality/std": 0.21573381125926971, "rewards/total_composite/mean": 0.5672909021377563, "rewards/total_composite/std": 0.13962553441524506, "reward": 0.5672909021377563, "reward_std": 0.13962554931640625, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15307047963142395, "sampling/sampling_logp_difference/max": 1.9560296535491943, "sampling/importance_sampling_ratio/min": 0.15023307502269745, "sampling/importance_sampling_ratio/mean": 1.0319684743881226, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8835284784436226, "clip_ratio/low_mean": 0.06878079008311033, "clip_ratio/low_min": 0.06878079008311033, "clip_ratio/high_mean": 0.07038584724068642, "clip_ratio/high_max": 0.07038584724068642, "clip_ratio/region_mean": 0.13916663732379675, "reward_total_mean": 0.5672909021377563, "reward_meter_mean": 0.5295940637588501, "reward_meter_std": 0.3171357214450836, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9773417115211487, "reward_repeat_soft_std": 0.015895428135991096, "reward_judge_quality_mean": 0.6737499833106995, "reward_judge_quality_std": 0.21573381125926971, "reward_total_composite_mean": 0.5672909021377563, "reward_total_composite_std": 0.13962553441524506} {"timestamp_utc": "2026-04-13T09:26:24Z", "mode": "train", "global_step": 832, "epoch": 0.08357609241587143, "loss": 0.0711, "grad_norm": 17.475072860717773, "learning_rate": 7.481818181818182e-06, "num_tokens": 1472318.0, "completions/mean_length": 24.875, "completions/min_length": 17.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.875, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.643103837966919, "rewards/meter/std": 0.37102043628692627, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.919399619102478, "rewards/repeat_soft/std": 0.05687469244003296, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.5491907596588135, "rewards/total_composite/std": 0.12342143803834915, "reward": 0.5491907596588135, "reward_std": 0.12342143803834915, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18710698187351227, "sampling/sampling_logp_difference/max": 2.4872217178344727, "sampling/importance_sampling_ratio/min": 0.08314063400030136, "sampling/importance_sampling_ratio/mean": 1.0076881647109985, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0633917525410652, "clip_ratio/low_mean": 0.06078296899795532, "clip_ratio/low_min": 0.06078296899795532, "clip_ratio/high_mean": 0.12785907462239265, "clip_ratio/high_max": 0.12785907462239265, "clip_ratio/region_mean": 0.18864204362034798, "reward_total_mean": 0.5491907596588135, "reward_meter_mean": 0.643103837966919, "reward_meter_std": 0.37102043628692627, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.919399619102478, "reward_repeat_soft_std": 0.05687469244003296, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.5491907596588135, "reward_total_composite_std": 0.12342143803834915} {"timestamp_utc": "2026-04-13T09:26:36Z", "mode": "train", "global_step": 833, "epoch": 0.08367654445002512, "loss": -0.1675, "grad_norm": 2.430192470550537, "learning_rate": 7.47878787878788e-06, "num_tokens": 1474137.0, "completions/mean_length": 189.375, "completions/min_length": 68.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 81.83333587646484, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 100.0, "rewards/meter/mean": 0.8023124933242798, "rewards/meter/std": 0.31727367639541626, "rewards/count_adherence/mean": 0.71875, "rewards/count_adherence/std": 0.24775780737400055, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.8893700242042542, "rewards/repeat_soft/std": 0.09991069883108139, "rewards/judge_quality/mean": 0.3137499988079071, "rewards/judge_quality/std": 0.17492344975471497, "rewards/total_composite/mean": 0.402348130941391, "rewards/total_composite/std": 0.25113871693611145, "reward": 0.402348130941391, "reward_std": 0.25113871693611145, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13309930264949799, "sampling/sampling_logp_difference/max": 3.3386030197143555, "sampling/importance_sampling_ratio/min": 0.03548649698495865, "sampling/importance_sampling_ratio/mean": 1.0186488628387451, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5958703383803368, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09196088742464781, "clip_ratio/high_max": 0.09196088742464781, "clip_ratio/region_mean": 0.09196088742464781, "reward_total_mean": 0.402348130941391, "reward_meter_mean": 0.8023124933242798, "reward_meter_std": 0.31727367639541626, "reward_count_adherence_mean": 0.71875, "reward_count_adherence_std": 0.24775780737400055, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.8893700242042542, "reward_repeat_soft_std": 0.09991069883108139, "reward_judge_quality_mean": 0.3137499988079071, "reward_judge_quality_std": 0.17492344975471497, "reward_total_composite_mean": 0.402348130941391, "reward_total_composite_std": 0.25113871693611145} {"timestamp_utc": "2026-04-13T09:26:43Z", "mode": "train", "global_step": 834, "epoch": 0.0837769964841788, "loss": 0.0629, "grad_norm": 11.07868480682373, "learning_rate": 7.4757575757575765e-06, "num_tokens": 1476039.0, "completions/mean_length": 61.75, "completions/min_length": 54.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.75, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.8380767107009888, "rewards/meter/std": 0.2552626430988312, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9660685658454895, "rewards/repeat_soft/std": 0.020650943741202354, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.10392305999994278, "rewards/total_composite/mean": 0.5503005981445312, "rewards/total_composite/std": 0.1013234481215477, "reward": 0.5503005981445312, "reward_std": 0.1013234481215477, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15240879356861115, "sampling/sampling_logp_difference/max": 2.377470016479492, "sampling/importance_sampling_ratio/min": 0.09278502315282822, "sampling/importance_sampling_ratio/mean": 0.9974220991134644, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8167963847517967, "clip_ratio/low_mean": 0.0525423726066947, "clip_ratio/low_min": 0.0525423726066947, "clip_ratio/high_mean": 0.08835308533161879, "clip_ratio/high_max": 0.08835308533161879, "clip_ratio/region_mean": 0.14089545793831348, "reward_total_mean": 0.5503005981445312, "reward_meter_mean": 0.8380767107009888, "reward_meter_std": 0.2552626430988312, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9660685658454895, "reward_repeat_soft_std": 0.020650943741202354, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.10392305999994278, "reward_total_composite_mean": 0.5503005981445312, "reward_total_composite_std": 0.1013234481215477} {"timestamp_utc": "2026-04-13T09:26:50Z", "mode": "train", "global_step": 835, "epoch": 0.0838774485183325, "loss": 0.0691, "grad_norm": 9.015896797180176, "learning_rate": 7.472727272727274e-06, "num_tokens": 1478109.0, "completions/mean_length": 90.75, "completions/min_length": 79.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.75, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.4803101718425751, "rewards/meter/std": 0.31699150800704956, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7889866232872009, "rewards/repeat_soft/std": 0.13812139630317688, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.10392305999994278, "rewards/total_composite/mean": 0.42656323313713074, "rewards/total_composite/std": 0.10647182166576385, "reward": 0.42656323313713074, "reward_std": 0.10647180676460266, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13574816286563873, "sampling/sampling_logp_difference/max": 1.7294883728027344, "sampling/importance_sampling_ratio/min": 0.17737513780593872, "sampling/importance_sampling_ratio/mean": 0.9962683916091919, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6161966919898987, "clip_ratio/low_mean": 0.06651968089863658, "clip_ratio/low_min": 0.06651968089863658, "clip_ratio/high_mean": 0.04569953680038452, "clip_ratio/high_max": 0.04569953680038452, "clip_ratio/region_mean": 0.1122192176990211, "reward_total_mean": 0.42656323313713074, "reward_meter_mean": 0.4803101718425751, "reward_meter_std": 0.31699150800704956, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7889866232872009, "reward_repeat_soft_std": 0.13812139630317688, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.10392305999994278, "reward_total_composite_mean": 0.42656323313713074, "reward_total_composite_std": 0.10647182166576385} {"timestamp_utc": "2026-04-13T09:26:56Z", "mode": "train", "global_step": 836, "epoch": 0.08397790055248619, "loss": 0.0204, "grad_norm": 15.014250755310059, "learning_rate": 7.46969696969697e-06, "num_tokens": 1479620.0, "completions/mean_length": 40.875, "completions/min_length": 36.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.875, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.24300700426101685, "rewards/meter/std": 0.28373533487319946, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9927163124084473, "rewards/repeat_soft/std": 0.01412123627960682, "rewards/judge_quality/mean": 0.48625001311302185, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.4226791262626648, "rewards/total_composite/std": 0.07671697437763214, "reward": 0.4226791262626648, "reward_std": 0.07671696692705154, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15632516145706177, "sampling/sampling_logp_difference/max": 2.2316935062408447, "sampling/importance_sampling_ratio/min": 0.10734648257493973, "sampling/importance_sampling_ratio/mean": 0.9939863681793213, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5575226359069347, "clip_ratio/low_mean": 0.09041997790336609, "clip_ratio/low_min": 0.09041997790336609, "clip_ratio/high_mean": 0.05626385845243931, "clip_ratio/high_max": 0.05626385845243931, "clip_ratio/region_mean": 0.1466838363558054, "reward_total_mean": 0.4226791262626648, "reward_meter_mean": 0.24300700426101685, "reward_meter_std": 0.28373533487319946, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9927163124084473, "reward_repeat_soft_std": 0.01412123627960682, "reward_judge_quality_mean": 0.48625001311302185, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.4226791262626648, "reward_total_composite_std": 0.07671697437763214} {"timestamp_utc": "2026-04-13T09:27:07Z", "mode": "train", "global_step": 837, "epoch": 0.08407835258663988, "loss": -0.0744, "grad_norm": 2.8593173027038574, "learning_rate": 7.4666666666666675e-06, "num_tokens": 1481119.0, "completions/mean_length": 86.375, "completions/min_length": 23.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 25.571430206298828, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.43489348888397217, "rewards/meter/std": 0.44809314608573914, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9363674521446228, "rewards/repeat_soft/std": 0.042148202657699585, "rewards/judge_quality/mean": 0.3712499737739563, "rewards/judge_quality/std": 0.1470119059085846, "rewards/total_composite/mean": 0.4052889943122864, "rewards/total_composite/std": 0.19342483580112457, "reward": 0.4052889943122864, "reward_std": 0.19342480599880219, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18456131219863892, "sampling/sampling_logp_difference/max": 1.0080249309539795, "sampling/importance_sampling_ratio/min": 0.36493903398513794, "sampling/importance_sampling_ratio/mean": 1.015906572341919, "sampling/importance_sampling_ratio/max": 1.7779897451400757, "entropy": 1.6088326200842857, "clip_ratio/low_mean": 0.03570930939167738, "clip_ratio/low_min": 0.03570930939167738, "clip_ratio/high_mean": 0.09555131569504738, "clip_ratio/high_max": 0.09555131569504738, "clip_ratio/region_mean": 0.13126062508672476, "reward_total_mean": 0.4052889943122864, "reward_meter_mean": 0.43489348888397217, "reward_meter_std": 0.44809314608573914, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9363674521446228, "reward_repeat_soft_std": 0.042148202657699585, "reward_judge_quality_mean": 0.3712499737739563, "reward_judge_quality_std": 0.1470119059085846, "reward_total_composite_mean": 0.4052889943122864, "reward_total_composite_std": 0.19342483580112457} {"timestamp_utc": "2026-04-13T09:27:18Z", "mode": "train", "global_step": 838, "epoch": 0.08417880462079357, "loss": -0.0852, "grad_norm": 2.8423261642456055, "learning_rate": 7.463636363636364e-06, "num_tokens": 1482522.0, "completions/mean_length": 87.375, "completions/min_length": 23.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 26.71428680419922, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.796506404876709, "rewards/meter/std": 0.35017162561416626, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9451595544815063, "rewards/repeat_soft/std": 0.03314792737364769, "rewards/judge_quality/mean": 0.3812499940395355, "rewards/judge_quality/std": 0.13452960550785065, "rewards/total_composite/mean": 0.49256080389022827, "rewards/total_composite/std": 0.22193673253059387, "reward": 0.49256080389022827, "reward_std": 0.22193673253059387, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16713367402553558, "sampling/sampling_logp_difference/max": 2.263957977294922, "sampling/importance_sampling_ratio/min": 0.1039382815361023, "sampling/importance_sampling_ratio/mean": 1.02517831325531, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0848853811621666, "clip_ratio/low_mean": 0.02500000037252903, "clip_ratio/low_min": 0.02500000037252903, "clip_ratio/high_mean": 0.11378003749996424, "clip_ratio/high_max": 0.11378003749996424, "clip_ratio/region_mean": 0.13878003787249327, "reward_total_mean": 0.49256080389022827, "reward_meter_mean": 0.796506404876709, "reward_meter_std": 0.35017162561416626, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9451595544815063, "reward_repeat_soft_std": 0.03314792737364769, "reward_judge_quality_mean": 0.3812499940395355, "reward_judge_quality_std": 0.13452960550785065, "reward_total_composite_mean": 0.49256080389022827, "reward_total_composite_std": 0.22193673253059387} {"timestamp_utc": "2026-04-13T09:27:29Z", "mode": "train", "global_step": 839, "epoch": 0.08427925665494726, "loss": -0.1898, "grad_norm": 3.640242338180542, "learning_rate": 7.460606060606061e-06, "num_tokens": 1484565.0, "completions/mean_length": 137.375, "completions/min_length": 69.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 83.85714721679688, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.32945507764816284, "rewards/meter/std": 0.2728482186794281, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.13363061845302582, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9720566272735596, "rewards/repeat_soft/std": 0.04077162221074104, "rewards/judge_quality/mean": 0.5325000286102295, "rewards/judge_quality/std": 0.22320717573165894, "rewards/total_composite/mean": 0.3884682059288025, "rewards/total_composite/std": 0.17261742055416107, "reward": 0.3884682059288025, "reward_std": 0.17261742055416107, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1431867629289627, "sampling/sampling_logp_difference/max": 1.8204381465911865, "sampling/importance_sampling_ratio/min": 0.16195477545261383, "sampling/importance_sampling_ratio/mean": 1.0115183591842651, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8769795000553131, "clip_ratio/low_mean": 0.035330988466739655, "clip_ratio/low_min": 0.035330988466739655, "clip_ratio/high_mean": 0.09252869058400393, "clip_ratio/high_max": 0.09252869058400393, "clip_ratio/region_mean": 0.12785967905074358, "reward_total_mean": 0.3884682059288025, "reward_meter_mean": 0.32945507764816284, "reward_meter_std": 0.2728482186794281, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.13363061845302582, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9720566272735596, "reward_repeat_soft_std": 0.04077162221074104, "reward_judge_quality_mean": 0.5325000286102295, "reward_judge_quality_std": 0.22320717573165894, "reward_total_composite_mean": 0.3884682059288025, "reward_total_composite_std": 0.17261742055416107} {"timestamp_utc": "2026-04-13T09:27:35Z", "mode": "train", "global_step": 840, "epoch": 0.08437970868910095, "loss": 0.0086, "grad_norm": 28.868541717529297, "learning_rate": 7.4575757575757575e-06, "num_tokens": 1485934.0, "completions/mean_length": 18.125, "completions/min_length": 13.0, "completions/max_length": 23.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 18.125, "completions/min_terminated_length": 13.0, "completions/max_terminated_length": 23.0, "rewards/meter/mean": 0.31399306654930115, "rewards/meter/std": 0.42830389738082886, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.22720351815223694, "rewards/total_composite/mean": 0.4486703872680664, "rewards/total_composite/std": 0.13358429074287415, "reward": 0.4486703872680664, "reward_std": 0.13358429074287415, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19885778427124023, "sampling/sampling_logp_difference/max": 1.3104972839355469, "sampling/importance_sampling_ratio/min": 0.26968589425086975, "sampling/importance_sampling_ratio/mean": 1.0157369375228882, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3750434964895248, "clip_ratio/low_mean": 0.1131312781944871, "clip_ratio/low_min": 0.1131312781944871, "clip_ratio/high_mean": 0.05902777798473835, "clip_ratio/high_max": 0.05902777798473835, "clip_ratio/region_mean": 0.17215905617922544, "reward_total_mean": 0.4486703872680664, "reward_meter_mean": 0.31399306654930115, "reward_meter_std": 0.42830389738082886, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.22720351815223694, "reward_total_composite_mean": 0.4486703872680664, "reward_total_composite_std": 0.13358429074287415} {"timestamp_utc": "2026-04-13T09:27:43Z", "mode": "train", "global_step": 841, "epoch": 0.08448016072325465, "loss": -0.0214, "grad_norm": 10.265496253967285, "learning_rate": 7.454545454545456e-06, "num_tokens": 1487932.0, "completions/mean_length": 71.75, "completions/min_length": 64.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.75, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.27685803174972534, "rewards/meter/std": 0.23641550540924072, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7995047569274902, "rewards/repeat_soft/std": 0.11472798138856888, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.3935416340827942, "rewards/total_composite/std": 0.06690897047519684, "reward": 0.3935416340827942, "reward_std": 0.06690897047519684, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12471328675746918, "sampling/sampling_logp_difference/max": 2.387075901031494, "sampling/importance_sampling_ratio/min": 0.09189800918102264, "sampling/importance_sampling_ratio/mean": 1.0074000358581543, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5988735593855381, "clip_ratio/low_mean": 0.0575826425338164, "clip_ratio/low_min": 0.0575826425338164, "clip_ratio/high_mean": 0.01458030566573143, "clip_ratio/high_max": 0.01458030566573143, "clip_ratio/region_mean": 0.07216294819954783, "reward_total_mean": 0.3935416340827942, "reward_meter_mean": 0.27685803174972534, "reward_meter_std": 0.23641550540924072, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7995047569274902, "reward_repeat_soft_std": 0.11472798138856888, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.3935416340827942, "reward_total_composite_std": 0.06690897047519684} {"timestamp_utc": "2026-04-13T09:27:55Z", "mode": "train", "global_step": 842, "epoch": 0.08458061275740834, "loss": -0.1792, "grad_norm": 2.87349271774292, "learning_rate": 7.451515151515152e-06, "num_tokens": 1490164.0, "completions/mean_length": 153.0, "completions/min_length": 89.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 101.71428680419922, "completions/min_terminated_length": 89.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.4708000421524048, "rewards/meter/std": 0.2996063232421875, "rewards/count_adherence/mean": 0.8541666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8525799512863159, "rewards/repeat_soft/std": 0.11474509537220001, "rewards/judge_quality/mean": 0.7075000405311584, "rewards/judge_quality/std": 0.29489707946777344, "rewards/total_composite/mean": 0.46092456579208374, "rewards/total_composite/std": 0.20109796524047852, "reward": 0.46092456579208374, "reward_std": 0.20109793543815613, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1308763027191162, "sampling/sampling_logp_difference/max": 1.7270450592041016, "sampling/importance_sampling_ratio/min": 0.17780904471874237, "sampling/importance_sampling_ratio/mean": 1.0077468156814575, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.48516032099723816, "clip_ratio/low_mean": 0.015625, "clip_ratio/low_min": 0.015625, "clip_ratio/high_mean": 0.10736322961747646, "clip_ratio/high_max": 0.10736322961747646, "clip_ratio/region_mean": 0.12298822961747646, "reward_total_mean": 0.46092456579208374, "reward_meter_mean": 0.4708000421524048, "reward_meter_std": 0.2996063232421875, "reward_count_adherence_mean": 0.8541666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8525799512863159, "reward_repeat_soft_std": 0.11474509537220001, "reward_judge_quality_mean": 0.7075000405311584, "reward_judge_quality_std": 0.29489707946777344, "reward_total_composite_mean": 0.46092456579208374, "reward_total_composite_std": 0.20109796524047852} {"timestamp_utc": "2026-04-13T09:28:02Z", "mode": "train", "global_step": 843, "epoch": 0.08468106479156202, "loss": 0.004, "grad_norm": 9.423258781433105, "learning_rate": 7.448484848484849e-06, "num_tokens": 1492334.0, "completions/mean_length": 79.25, "completions/min_length": 59.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.25, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.7220500111579895, "rewards/meter/std": 0.34564322233200073, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9802860021591187, "rewards/repeat_soft/std": 0.0205127764493227, "rewards/judge_quality/mean": 0.5487499833106995, "rewards/judge_quality/std": 0.22937417030334473, "rewards/total_composite/mean": 0.5785923600196838, "rewards/total_composite/std": 0.19729620218276978, "reward": 0.5785923600196838, "reward_std": 0.19729618728160858, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13742201030254364, "sampling/sampling_logp_difference/max": 1.9483990669250488, "sampling/importance_sampling_ratio/min": 0.14250202476978302, "sampling/importance_sampling_ratio/mean": 1.0099302530288696, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7041908577084541, "clip_ratio/low_mean": 0.08119471371173859, "clip_ratio/low_min": 0.08119471371173859, "clip_ratio/high_mean": 0.03453873749822378, "clip_ratio/high_max": 0.03453873749822378, "clip_ratio/region_mean": 0.11573345120996237, "reward_total_mean": 0.5785923600196838, "reward_meter_mean": 0.7220500111579895, "reward_meter_std": 0.34564322233200073, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9802860021591187, "reward_repeat_soft_std": 0.0205127764493227, "reward_judge_quality_mean": 0.5487499833106995, "reward_judge_quality_std": 0.22937417030334473, "reward_total_composite_mean": 0.5785923600196838, "reward_total_composite_std": 0.19729620218276978} {"timestamp_utc": "2026-04-13T09:28:08Z", "mode": "train", "global_step": 844, "epoch": 0.08478151682571572, "loss": 0.0033, "grad_norm": 13.6819486618042, "learning_rate": 7.445454545454546e-06, "num_tokens": 1494315.0, "completions/mean_length": 73.625, "completions/min_length": 57.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.625, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.19047191739082336, "rewards/meter/std": 0.254278302192688, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9960660934448242, "rewards/repeat_soft/std": 0.0026364249642938375, "rewards/judge_quality/mean": 0.5987499952316284, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.40190425515174866, "rewards/total_composite/std": 0.15848101675510406, "reward": 0.40190425515174866, "reward_std": 0.15848100185394287, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1798246055841446, "sampling/sampling_logp_difference/max": 3.6892249584198, "sampling/importance_sampling_ratio/min": 0.024991365149617195, "sampling/importance_sampling_ratio/mean": 1.0255519151687622, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0761237666010857, "clip_ratio/low_mean": 0.1708741346374154, "clip_ratio/low_min": 0.1708741346374154, "clip_ratio/high_mean": 0.0234375, "clip_ratio/high_max": 0.0234375, "clip_ratio/region_mean": 0.1943116346374154, "reward_total_mean": 0.40190425515174866, "reward_meter_mean": 0.19047191739082336, "reward_meter_std": 0.254278302192688, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9960660934448242, "reward_repeat_soft_std": 0.0026364249642938375, "reward_judge_quality_mean": 0.5987499952316284, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.40190425515174866, "reward_total_composite_std": 0.15848101675510406} {"timestamp_utc": "2026-04-13T09:28:14Z", "mode": "train", "global_step": 845, "epoch": 0.08488196885986941, "loss": 0.0103, "grad_norm": 12.717124938964844, "learning_rate": 7.442424242424243e-06, "num_tokens": 1495911.0, "completions/mean_length": 39.5, "completions/min_length": 34.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.5, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.9568942785263062, "rewards/meter/std": 0.049314092844724655, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9216736555099487, "rewards/repeat_soft/std": 0.11439024657011032, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.6089969873428345, "rewards/total_composite/std": 0.019405582919716835, "reward": 0.6089969873428345, "reward_std": 0.019405582919716835, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17396031320095062, "sampling/sampling_logp_difference/max": 1.5292034149169922, "sampling/importance_sampling_ratio/min": 0.216708242893219, "sampling/importance_sampling_ratio/mean": 0.9948079586029053, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.194935329258442, "clip_ratio/low_mean": 0.03553921636193991, "clip_ratio/low_min": 0.03553921636193991, "clip_ratio/high_mean": 0.14455649629235268, "clip_ratio/high_max": 0.14455649629235268, "clip_ratio/region_mean": 0.18009571265429258, "reward_total_mean": 0.6089969873428345, "reward_meter_mean": 0.9568942785263062, "reward_meter_std": 0.049314092844724655, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9216736555099487, "reward_repeat_soft_std": 0.11439024657011032, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.6089969873428345, "reward_total_composite_std": 0.019405582919716835} {"timestamp_utc": "2026-04-13T09:28:25Z", "mode": "train", "global_step": 846, "epoch": 0.08498242089402311, "loss": -0.1199, "grad_norm": 3.850022554397583, "learning_rate": 7.439393939393939e-06, "num_tokens": 1497555.0, "completions/mean_length": 101.5, "completions/min_length": 37.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 42.85714340209961, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.33093252778053284, "rewards/meter/std": 0.307478129863739, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9527453184127808, "rewards/repeat_soft/std": 0.09837198257446289, "rewards/judge_quality/mean": 0.5437500476837158, "rewards/judge_quality/std": 0.29861289262771606, "rewards/total_composite/mean": 0.4027683138847351, "rewards/total_composite/std": 0.18549330532550812, "reward": 0.4027683138847351, "reward_std": 0.18549330532550812, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17301549017429352, "sampling/sampling_logp_difference/max": 1.2793481349945068, "sampling/importance_sampling_ratio/min": 0.32019028067588806, "sampling/importance_sampling_ratio/mean": 1.0440881252288818, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0814722403883934, "clip_ratio/low_mean": 0.030868902802467346, "clip_ratio/low_min": 0.030868902802467346, "clip_ratio/high_mean": 0.1328445840626955, "clip_ratio/high_max": 0.1328445840626955, "clip_ratio/region_mean": 0.16371348686516285, "reward_total_mean": 0.4027683138847351, "reward_meter_mean": 0.33093252778053284, "reward_meter_std": 0.307478129863739, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9527453184127808, "reward_repeat_soft_std": 0.09837198257446289, "reward_judge_quality_mean": 0.5437500476837158, "reward_judge_quality_std": 0.29861289262771606, "reward_total_composite_mean": 0.4027683138847351, "reward_total_composite_std": 0.18549330532550812} {"timestamp_utc": "2026-04-13T09:28:32Z", "mode": "train", "global_step": 847, "epoch": 0.08508287292817679, "loss": 0.0692, "grad_norm": 14.30763053894043, "learning_rate": 7.4363636363636375e-06, "num_tokens": 1499209.0, "completions/mean_length": 40.75, "completions/min_length": 36.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.75, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.2561221420764923, "rewards/meter/std": 0.38831233978271484, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9554833769798279, "rewards/repeat_soft/std": 0.04576227441430092, "rewards/judge_quality/mean": 0.5049999952316284, "rewards/judge_quality/std": 0.11686378717422485, "rewards/total_composite/mean": 0.4316032826900482, "rewards/total_composite/std": 0.13052450120449066, "reward": 0.4316032826900482, "reward_std": 0.13052450120449066, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15778307616710663, "sampling/sampling_logp_difference/max": 2.067727565765381, "sampling/importance_sampling_ratio/min": 0.1264728456735611, "sampling/importance_sampling_ratio/mean": 1.0026819705963135, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0495265498757362, "clip_ratio/low_mean": 0.11036532931029797, "clip_ratio/low_min": 0.11036532931029797, "clip_ratio/high_mean": 0.04537786729633808, "clip_ratio/high_max": 0.04537786729633808, "clip_ratio/region_mean": 0.15574319660663605, "reward_total_mean": 0.4316032826900482, "reward_meter_mean": 0.2561221420764923, "reward_meter_std": 0.38831233978271484, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9554833769798279, "reward_repeat_soft_std": 0.04576227441430092, "reward_judge_quality_mean": 0.5049999952316284, "reward_judge_quality_std": 0.11686378717422485, "reward_total_composite_mean": 0.4316032826900482, "reward_total_composite_std": 0.13052450120449066} {"timestamp_utc": "2026-04-13T09:28:43Z", "mode": "train", "global_step": 848, "epoch": 0.08518332496233048, "loss": -0.0575, "grad_norm": 1.3515770435333252, "learning_rate": 7.433333333333334e-06, "num_tokens": 1500625.0, "completions/mean_length": 274.0, "completions/min_length": 33.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.5, "completions/mean_terminated_length": 36.0, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.1757933795452118, "rewards/meter/std": 0.11196483671665192, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.5, "rewards/hard_gate/std": 0.5345224738121033, "rewards/repeat_soft/mean": 0.9779536724090576, "rewards/repeat_soft/std": 0.036843180656433105, "rewards/judge_quality/mean": 0.2175000011920929, "rewards/judge_quality/std": 0.18873639404773712, "rewards/total_composite/mean": 0.20365145802497864, "rewards/total_composite/std": 0.21790151298046112, "reward": 0.20365145802497864, "reward_std": 0.21790151298046112, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18480101227760315, "sampling/sampling_logp_difference/max": 1.077570915222168, "sampling/importance_sampling_ratio/min": 0.3404214084148407, "sampling/importance_sampling_ratio/mean": 0.9984260201454163, "sampling/importance_sampling_ratio/max": 1.741991639137268, "entropy": 0.7906372845172882, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.08901694789528847, "clip_ratio/high_max": 0.08901694789528847, "clip_ratio/region_mean": 0.08901694789528847, "reward_total_mean": 0.20365145802497864, "reward_meter_mean": 0.1757933795452118, "reward_meter_std": 0.11196483671665192, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.5, "reward_hard_gate_std": 0.5345224738121033, "reward_repeat_soft_mean": 0.9779536724090576, "reward_repeat_soft_std": 0.036843180656433105, "reward_judge_quality_mean": 0.2175000011920929, "reward_judge_quality_std": 0.18873639404773712, "reward_total_composite_mean": 0.20365145802497864, "reward_total_composite_std": 0.21790151298046112} {"timestamp_utc": "2026-04-13T09:28:49Z", "mode": "train", "global_step": 849, "epoch": 0.08528377699648418, "loss": 0.017, "grad_norm": 17.583232879638672, "learning_rate": 7.430303030303031e-06, "num_tokens": 1502151.0, "completions/mean_length": 36.75, "completions/min_length": 30.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.75, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.34428954124450684, "rewards/meter/std": 0.35059359669685364, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9794764518737793, "rewards/repeat_soft/std": 0.013408654369413853, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.17136012017726898, "rewards/total_composite/mean": 0.4079633355140686, "rewards/total_composite/std": 0.18753710389137268, "reward": 0.4079633355140686, "reward_std": 0.18753710389137268, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1670643538236618, "sampling/sampling_logp_difference/max": 2.1720149517059326, "sampling/importance_sampling_ratio/min": 0.1139477863907814, "sampling/importance_sampling_ratio/mean": 1.011920690536499, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8922168835997581, "clip_ratio/low_mean": 0.026709401980042458, "clip_ratio/low_min": 0.026709401980042458, "clip_ratio/high_mean": 0.11197670269757509, "clip_ratio/high_max": 0.11197670269757509, "clip_ratio/region_mean": 0.13868610467761755, "reward_total_mean": 0.4079633355140686, "reward_meter_mean": 0.34428954124450684, "reward_meter_std": 0.35059359669685364, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9794764518737793, "reward_repeat_soft_std": 0.013408654369413853, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.17136012017726898, "reward_total_composite_mean": 0.4079633355140686, "reward_total_composite_std": 0.18753710389137268} {"timestamp_utc": "2026-04-13T09:28:55Z", "mode": "train", "global_step": 850, "epoch": 0.08538422903063787, "loss": 0.0287, "grad_norm": 13.789236068725586, "learning_rate": 7.4272727272727275e-06, "num_tokens": 1503640.0, "completions/mean_length": 38.125, "completions/min_length": 34.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.125, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.22956950962543488, "rewards/meter/std": 0.20842336118221283, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9453689455986023, "rewards/repeat_soft/std": 0.03891696780920029, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.22720351815223694, "rewards/total_composite/mean": 0.4220011532306671, "rewards/total_composite/std": 0.06589780002832413, "reward": 0.4220011532306671, "reward_std": 0.06589780747890472, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13908246159553528, "sampling/sampling_logp_difference/max": 1.6850318908691406, "sampling/importance_sampling_ratio/min": 0.18543851375579834, "sampling/importance_sampling_ratio/mean": 1.0063995122909546, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9740190431475639, "clip_ratio/low_mean": 0.065509426407516, "clip_ratio/low_min": 0.065509426407516, "clip_ratio/high_mean": 0.062609649496153, "clip_ratio/high_max": 0.062609649496153, "clip_ratio/region_mean": 0.128119075903669, "reward_total_mean": 0.4220011532306671, "reward_meter_mean": 0.22956950962543488, "reward_meter_std": 0.20842336118221283, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9453689455986023, "reward_repeat_soft_std": 0.03891696780920029, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.22720351815223694, "reward_total_composite_mean": 0.4220011532306671, "reward_total_composite_std": 0.06589780002832413} {"timestamp_utc": "2026-04-13T09:30:07Z", "mode": "eval", "global_step": 850, "epoch": 0.08538422903063787, "eval_loss": NaN, "eval_runtime": 71.8131, "eval_samples_per_second": 1.114, "eval_steps_per_second": 0.139, "eval_num_tokens": 1503640.0, "eval_completions/mean_length": 124.15, "eval_completions/min_length": 32.8, "eval_completions/max_length": 385.3, "eval_completions/clipped_ratio": 0.1375, "eval_completions/mean_terminated_length": 62.55821685791015, "eval_completions/min_terminated_length": 32.8, "eval_completions/max_terminated_length": 97.5, "eval_rewards/meter/mean": 0.5041633427143097, "eval_rewards/meter/std": 0.3839849755167961, "eval_rewards/count_adherence/mean": 0.8604166507720947, "eval_rewards/count_adherence/std": 0.1260624572634697, "eval_rewards/hard_gate/mean": 0.8625, "eval_rewards/hard_gate/std": 0.28575828671455383, "eval_rewards/repeat_soft/mean": 0.9330996811389923, "eval_rewards/repeat_soft/std": 0.10413262154906988, "eval_rewards/judge_quality/mean": 0.47837499976158143, "eval_rewards/judge_quality/std": 0.23521116748452187, "eval_rewards/total_composite/mean": 0.43147261142730714, "eval_rewards/total_composite/std": 0.2192233495414257, "eval_reward": 0.43147261142730714, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.08082602322101592, "eval_sampling/sampling_logp_difference/max": 0.9368991374969482, "eval_sampling/importance_sampling_ratio/min": 0.3969268500804901, "eval_sampling/importance_sampling_ratio/mean": 1.0231210112571716, "eval_sampling/importance_sampling_ratio/max": 1.523264479637146, "eval_entropy": 0.988732784986496, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.43147261142730714, "eval_reward_meter_mean": 0.5041633427143097, "eval_reward_meter_std": 0.3839849755167961, "eval_reward_count_adherence_mean": 0.8604166507720947, "eval_reward_count_adherence_std": 0.1260624572634697, "eval_reward_hard_gate_mean": 0.8625, "eval_reward_hard_gate_std": 0.28575828671455383, "eval_reward_repeat_soft_mean": 0.9330996811389923, "eval_reward_repeat_soft_std": 0.10413262154906988, "eval_reward_judge_quality_mean": 0.47837499976158143, "eval_reward_judge_quality_std": 0.23521116748452187, "eval_reward_total_composite_mean": 0.43147261142730714, "eval_reward_total_composite_std": 0.2192233495414257} {"timestamp_utc": "2026-04-13T09:30:17Z", "mode": "train", "global_step": 851, "epoch": 0.08548468106479157, "loss": 0.1141, "grad_norm": 14.212329864501953, "learning_rate": 7.424242424242425e-06, "num_tokens": 1505291.0, "completions/mean_length": 44.375, "completions/min_length": 37.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.375, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.5698230266571045, "rewards/meter/std": 0.3540171980857849, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.973199725151062, "rewards/repeat_soft/std": 0.015333795920014381, "rewards/judge_quality/mean": 0.8524999618530273, "rewards/judge_quality/std": 0.17531605064868927, "rewards/total_composite/mean": 0.6624869108200073, "rewards/total_composite/std": 0.2213582694530487, "reward": 0.6624869108200073, "reward_std": 0.2213582694530487, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14528997242450714, "sampling/sampling_logp_difference/max": 1.5352954864501953, "sampling/importance_sampling_ratio/min": 0.21539203822612762, "sampling/importance_sampling_ratio/mean": 1.0009996891021729, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.897915206849575, "clip_ratio/low_mean": 0.07398198451846838, "clip_ratio/low_min": 0.07398198451846838, "clip_ratio/high_mean": 0.04296270105987787, "clip_ratio/high_max": 0.04296270105987787, "clip_ratio/region_mean": 0.11694468557834625, "reward_total_mean": 0.6624869108200073, "reward_meter_mean": 0.5698230266571045, "reward_meter_std": 0.3540171980857849, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.973199725151062, "reward_repeat_soft_std": 0.015333795920014381, "reward_judge_quality_mean": 0.8524999618530273, "reward_judge_quality_std": 0.17531605064868927, "reward_total_composite_mean": 0.6624869108200073, "reward_total_composite_std": 0.2213582694530487} {"timestamp_utc": "2026-04-13T09:30:29Z", "mode": "train", "global_step": 852, "epoch": 0.08558513309894525, "loss": -0.1035, "grad_norm": 3.831021547317505, "learning_rate": 7.421212121212121e-06, "num_tokens": 1507089.0, "completions/mean_length": 109.75, "completions/min_length": 40.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 52.28571701049805, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.758591890335083, "rewards/meter/std": 0.20175181329250336, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.13363061845302582, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9904145002365112, "rewards/repeat_soft/std": 0.0077845207415521145, "rewards/judge_quality/mean": 0.6362500190734863, "rewards/judge_quality/std": 0.29635587334632874, "rewards/total_composite/mean": 0.6015548706054688, "rewards/total_composite/std": 0.27526721358299255, "reward": 0.6015548706054688, "reward_std": 0.27526721358299255, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15900366008281708, "sampling/sampling_logp_difference/max": 2.1218101978302, "sampling/importance_sampling_ratio/min": 0.11981455236673355, "sampling/importance_sampling_ratio/mean": 1.0003243684768677, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6033063903450966, "clip_ratio/low_mean": 0.06447595916688442, "clip_ratio/low_min": 0.06447595916688442, "clip_ratio/high_mean": 0.07329545635730028, "clip_ratio/high_max": 0.07329545635730028, "clip_ratio/region_mean": 0.1377714155241847, "reward_total_mean": 0.6015548706054688, "reward_meter_mean": 0.758591890335083, "reward_meter_std": 0.20175181329250336, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.13363061845302582, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9904145002365112, "reward_repeat_soft_std": 0.0077845207415521145, "reward_judge_quality_mean": 0.6362500190734863, "reward_judge_quality_std": 0.29635587334632874, "reward_total_composite_mean": 0.6015548706054688, "reward_total_composite_std": 0.27526721358299255} {"timestamp_utc": "2026-04-13T09:30:40Z", "mode": "train", "global_step": 853, "epoch": 0.08568558513309894, "loss": -0.0834, "grad_norm": 1.662527322769165, "learning_rate": 7.4181818181818185e-06, "num_tokens": 1508481.0, "completions/mean_length": 155.0, "completions/min_length": 29.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 36.0, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.2141370326280594, "rewards/meter/std": 0.3055233061313629, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.3720119297504425, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9991523027420044, "rewards/repeat_soft/std": 0.0015711163869127631, "rewards/judge_quality/mean": 0.5774999856948853, "rewards/judge_quality/std": 0.3919456899166107, "rewards/total_composite/mean": 0.3827788233757019, "rewards/total_composite/std": 0.29095616936683655, "reward": 0.3827788233757019, "reward_std": 0.29095613956451416, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1022883877158165, "sampling/sampling_logp_difference/max": 1.447120189666748, "sampling/importance_sampling_ratio/min": 0.23524677753448486, "sampling/importance_sampling_ratio/mean": 0.9997597336769104, "sampling/importance_sampling_ratio/max": 1.6579594612121582, "entropy": 0.4650096446275711, "clip_ratio/low_mean": 0.01993534481152892, "clip_ratio/low_min": 0.01993534481152892, "clip_ratio/high_mean": 0.06936437916010618, "clip_ratio/high_max": 0.06936437916010618, "clip_ratio/region_mean": 0.0892997239716351, "reward_total_mean": 0.3827788233757019, "reward_meter_mean": 0.2141370326280594, "reward_meter_std": 0.3055233061313629, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.3720119297504425, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9991523027420044, "reward_repeat_soft_std": 0.0015711163869127631, "reward_judge_quality_mean": 0.5774999856948853, "reward_judge_quality_std": 0.3919456899166107, "reward_total_composite_mean": 0.3827788233757019, "reward_total_composite_std": 0.29095616936683655} {"timestamp_utc": "2026-04-13T09:30:46Z", "mode": "train", "global_step": 854, "epoch": 0.08578603716725264, "loss": 0.1215, "grad_norm": 20.572551727294922, "learning_rate": 7.415151515151515e-06, "num_tokens": 1509884.0, "completions/mean_length": 22.375, "completions/min_length": 17.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.375, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.6981173753738403, "rewards/meter/std": 0.4420422911643982, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8873992562294006, "rewards/repeat_soft/std": 0.11063570529222488, "rewards/judge_quality/mean": 0.6775000095367432, "rewards/judge_quality/std": 0.25949129462242126, "rewards/total_composite/mean": 0.6501049995422363, "rewards/total_composite/std": 0.25837576389312744, "reward": 0.6501049995422363, "reward_std": 0.25837576389312744, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18870551884174347, "sampling/sampling_logp_difference/max": 1.3879594802856445, "sampling/importance_sampling_ratio/min": 0.24958406388759613, "sampling/importance_sampling_ratio/mean": 1.016324758529663, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5916697159409523, "clip_ratio/low_mean": 0.07532964693382382, "clip_ratio/low_min": 0.07532964693382382, "clip_ratio/high_mean": 0.07359601557254791, "clip_ratio/high_max": 0.07359601557254791, "clip_ratio/region_mean": 0.14892566250637174, "reward_total_mean": 0.6501049995422363, "reward_meter_mean": 0.6981173753738403, "reward_meter_std": 0.4420422911643982, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8873992562294006, "reward_repeat_soft_std": 0.11063570529222488, "reward_judge_quality_mean": 0.6775000095367432, "reward_judge_quality_std": 0.25949129462242126, "reward_total_composite_mean": 0.6501049995422363, "reward_total_composite_std": 0.25837576389312744} {"timestamp_utc": "2026-04-13T09:30:58Z", "mode": "train", "global_step": 855, "epoch": 0.08588648920140633, "loss": -0.09, "grad_norm": 2.11004376411438, "learning_rate": 7.412121212121213e-06, "num_tokens": 1511385.0, "completions/mean_length": 160.625, "completions/min_length": 34.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 43.5, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.4706278443336487, "rewards/meter/std": 0.3859804570674896, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.25877460837364197, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9950848817825317, "rewards/repeat_soft/std": 0.007068851962685585, "rewards/judge_quality/mean": 0.5062500238418579, "rewards/judge_quality/std": 0.32407838106155396, "rewards/total_composite/mean": 0.4136121869087219, "rewards/total_composite/std": 0.29443883895874023, "reward": 0.4136121869087219, "reward_std": 0.29443883895874023, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17346595227718353, "sampling/sampling_logp_difference/max": 1.8882384300231934, "sampling/importance_sampling_ratio/min": 0.15133817493915558, "sampling/importance_sampling_ratio/mean": 1.0436005592346191, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1971502006053925, "clip_ratio/low_mean": 0.05434782616794109, "clip_ratio/low_min": 0.05434782616794109, "clip_ratio/high_mean": 0.0963926874101162, "clip_ratio/high_max": 0.0963926874101162, "clip_ratio/region_mean": 0.1507405135780573, "reward_total_mean": 0.4136121869087219, "reward_meter_mean": 0.4706278443336487, "reward_meter_std": 0.3859804570674896, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.25877460837364197, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9950848817825317, "reward_repeat_soft_std": 0.007068851962685585, "reward_judge_quality_mean": 0.5062500238418579, "reward_judge_quality_std": 0.32407838106155396, "reward_total_composite_mean": 0.4136121869087219, "reward_total_composite_std": 0.29443883895874023} {"timestamp_utc": "2026-04-13T09:31:09Z", "mode": "train", "global_step": 856, "epoch": 0.08598694123556001, "loss": -0.0471, "grad_norm": 5.2753214836120605, "learning_rate": 7.40909090909091e-06, "num_tokens": 1512967.0, "completions/mean_length": 96.75, "completions/min_length": 34.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 37.42857360839844, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.4234778583049774, "rewards/meter/std": 0.45958077907562256, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8645415306091309, "rewards/repeat_soft/std": 0.08556653559207916, "rewards/judge_quality/mean": 0.5275000333786011, "rewards/judge_quality/std": 0.3520044684410095, "rewards/total_composite/mean": 0.4821651577949524, "rewards/total_composite/std": 0.318382203578949, "reward": 0.4821651577949524, "reward_std": 0.318382203578949, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1610652208328247, "sampling/sampling_logp_difference/max": 1.7582321166992188, "sampling/importance_sampling_ratio/min": 0.17234928905963898, "sampling/importance_sampling_ratio/mean": 0.969700276851654, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7111929133534431, "clip_ratio/low_mean": 0.048448143526911736, "clip_ratio/low_min": 0.048448143526911736, "clip_ratio/high_mean": 0.05237358249723911, "clip_ratio/high_max": 0.05237358249723911, "clip_ratio/region_mean": 0.10082172602415085, "reward_total_mean": 0.4821651577949524, "reward_meter_mean": 0.4234778583049774, "reward_meter_std": 0.45958077907562256, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8645415306091309, "reward_repeat_soft_std": 0.08556653559207916, "reward_judge_quality_mean": 0.5275000333786011, "reward_judge_quality_std": 0.3520044684410095, "reward_total_composite_mean": 0.4821651577949524, "reward_total_composite_std": 0.318382203578949} {"timestamp_utc": "2026-04-13T09:31:21Z", "mode": "train", "global_step": 857, "epoch": 0.08608739326971371, "loss": -0.1533, "grad_norm": 2.2963130474090576, "learning_rate": 7.406060606060607e-06, "num_tokens": 1514965.0, "completions/mean_length": 250.75, "completions/min_length": 82.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 94.0, "completions/min_terminated_length": 82.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.7481331825256348, "rewards/meter/std": 0.23359724879264832, "rewards/count_adherence/mean": 0.7749999761581421, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.8709614872932434, "rewards/repeat_soft/std": 0.10680233687162399, "rewards/judge_quality/mean": 0.38999998569488525, "rewards/judge_quality/std": 0.24686609208583832, "rewards/total_composite/mean": 0.3642788231372833, "rewards/total_composite/std": 0.32161790132522583, "reward": 0.3642788231372833, "reward_std": 0.32161790132522583, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14991313219070435, "sampling/sampling_logp_difference/max": 1.635301113128662, "sampling/importance_sampling_ratio/min": 0.19489367306232452, "sampling/importance_sampling_ratio/mean": 1.0165810585021973, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6166017353534698, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.07782910205423832, "clip_ratio/high_max": 0.07782910205423832, "clip_ratio/region_mean": 0.07782910205423832, "reward_total_mean": 0.3642788231372833, "reward_meter_mean": 0.7481331825256348, "reward_meter_std": 0.23359724879264832, "reward_count_adherence_mean": 0.7749999761581421, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.8709614872932434, "reward_repeat_soft_std": 0.10680233687162399, "reward_judge_quality_mean": 0.38999998569488525, "reward_judge_quality_std": 0.24686609208583832, "reward_total_composite_mean": 0.3642788231372833, "reward_total_composite_std": 0.32161790132522583} {"timestamp_utc": "2026-04-13T09:31:27Z", "mode": "train", "global_step": 858, "epoch": 0.0861878453038674, "loss": 0.0507, "grad_norm": 10.526352882385254, "learning_rate": 7.403030303030304e-06, "num_tokens": 1516484.0, "completions/mean_length": 44.875, "completions/min_length": 38.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.875, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.5815331935882568, "rewards/meter/std": 0.3391035795211792, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9906216263771057, "rewards/repeat_soft/std": 0.008486546576023102, "rewards/judge_quality/mean": 0.6412500143051147, "rewards/judge_quality/std": 0.3280869722366333, "rewards/total_composite/mean": 0.5933663249015808, "rewards/total_composite/std": 0.22715814411640167, "reward": 0.5933663249015808, "reward_std": 0.22715814411640167, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14228667318820953, "sampling/sampling_logp_difference/max": 2.4088973999023438, "sampling/importance_sampling_ratio/min": 0.08991438150405884, "sampling/importance_sampling_ratio/mean": 1.0182994604110718, "sampling/importance_sampling_ratio/max": 1.898579716682434, "entropy": 0.9067439511418343, "clip_ratio/low_mean": 0.07189829740673304, "clip_ratio/low_min": 0.07189829740673304, "clip_ratio/high_mean": 0.039669995196163654, "clip_ratio/high_max": 0.039669995196163654, "clip_ratio/region_mean": 0.11156829260289669, "reward_total_mean": 0.5933663249015808, "reward_meter_mean": 0.5815331935882568, "reward_meter_std": 0.3391035795211792, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9906216263771057, "reward_repeat_soft_std": 0.008486546576023102, "reward_judge_quality_mean": 0.6412500143051147, "reward_judge_quality_std": 0.3280869722366333, "reward_total_composite_mean": 0.5933663249015808, "reward_total_composite_std": 0.22715814411640167} {"timestamp_utc": "2026-04-13T09:31:39Z", "mode": "train", "global_step": 859, "epoch": 0.0862882973380211, "loss": -0.0944, "grad_norm": 1.9822921752929688, "learning_rate": 7.4e-06, "num_tokens": 1517939.0, "completions/mean_length": 156.875, "completions/min_length": 30.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 38.5, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.5484725832939148, "rewards/meter/std": 0.34517568349838257, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9916176199913025, "rewards/repeat_soft/std": 0.014756565913558006, "rewards/judge_quality/mean": 0.38999998569488525, "rewards/judge_quality/std": 0.27166154980659485, "rewards/total_composite/mean": 0.3975730538368225, "rewards/total_composite/std": 0.25572142004966736, "reward": 0.3975730538368225, "reward_std": 0.25572139024734497, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15622539818286896, "sampling/sampling_logp_difference/max": 1.3402645587921143, "sampling/importance_sampling_ratio/min": 0.2617764174938202, "sampling/importance_sampling_ratio/mean": 1.0125941038131714, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7355927750468254, "clip_ratio/low_mean": 0.0234375, "clip_ratio/low_min": 0.0234375, "clip_ratio/high_mean": 0.08745039720088243, "clip_ratio/high_max": 0.08745039720088243, "clip_ratio/region_mean": 0.11088789720088243, "reward_total_mean": 0.3975730538368225, "reward_meter_mean": 0.5484725832939148, "reward_meter_std": 0.34517568349838257, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9916176199913025, "reward_repeat_soft_std": 0.014756565913558006, "reward_judge_quality_mean": 0.38999998569488525, "reward_judge_quality_std": 0.27166154980659485, "reward_total_composite_mean": 0.3975730538368225, "reward_total_composite_std": 0.25572142004966736} {"timestamp_utc": "2026-04-13T09:31:46Z", "mode": "train", "global_step": 860, "epoch": 0.08638874937217479, "loss": 0.0341, "grad_norm": 13.615326881408691, "learning_rate": 7.396969696969698e-06, "num_tokens": 1519874.0, "completions/mean_length": 66.875, "completions/min_length": 56.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.875, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.8078511953353882, "rewards/meter/std": 0.31829094886779785, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9427466988563538, "rewards/repeat_soft/std": 0.03384443372488022, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.6276594400405884, "rewards/total_composite/std": 0.2029723823070526, "reward": 0.6276594400405884, "reward_std": 0.20297236740589142, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15863348543643951, "sampling/sampling_logp_difference/max": 1.6779999732971191, "sampling/importance_sampling_ratio/min": 0.18674710392951965, "sampling/importance_sampling_ratio/mean": 1.0145034790039062, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2692635208368301, "clip_ratio/low_mean": 0.11109228897839785, "clip_ratio/low_min": 0.11109228897839785, "clip_ratio/high_mean": 0.07190890610218048, "clip_ratio/high_max": 0.07190890610218048, "clip_ratio/region_mean": 0.18300119508057833, "reward_total_mean": 0.6276594400405884, "reward_meter_mean": 0.8078511953353882, "reward_meter_std": 0.31829094886779785, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9427466988563538, "reward_repeat_soft_std": 0.03384443372488022, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.6276594400405884, "reward_total_composite_std": 0.2029723823070526} {"timestamp_utc": "2026-04-13T09:31:57Z", "mode": "train", "global_step": 861, "epoch": 0.08648920140632847, "loss": -0.1475, "grad_norm": 3.81089448928833, "learning_rate": 7.393939393939395e-06, "num_tokens": 1522021.0, "completions/mean_length": 155.375, "completions/min_length": 89.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 104.42857360839844, "completions/min_terminated_length": 89.0, "completions/max_terminated_length": 118.0, "rewards/meter/mean": 0.4980456233024597, "rewards/meter/std": 0.23626483976840973, "rewards/count_adherence/mean": 0.7916666269302368, "rewards/count_adherence/std": 0.07715165615081787, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9310250282287598, "rewards/repeat_soft/std": 0.04550262540578842, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.2386719286441803, "rewards/total_composite/mean": 0.4168010950088501, "rewards/total_composite/std": 0.19585895538330078, "reward": 0.4168010950088501, "reward_std": 0.1958589404821396, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14212839305400848, "sampling/sampling_logp_difference/max": 2.3954758644104004, "sampling/importance_sampling_ratio/min": 0.09112930297851562, "sampling/importance_sampling_ratio/mean": 1.0250738859176636, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9451244100928307, "clip_ratio/low_mean": 0.026294946670532227, "clip_ratio/low_min": 0.026294946670532227, "clip_ratio/high_mean": 0.07693687546998262, "clip_ratio/high_max": 0.07693687546998262, "clip_ratio/region_mean": 0.10323182214051485, "reward_total_mean": 0.4168010950088501, "reward_meter_mean": 0.4980456233024597, "reward_meter_std": 0.23626483976840973, "reward_count_adherence_mean": 0.7916666269302368, "reward_count_adherence_std": 0.07715165615081787, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9310250282287598, "reward_repeat_soft_std": 0.04550262540578842, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.2386719286441803, "reward_total_composite_mean": 0.4168010950088501, "reward_total_composite_std": 0.19585895538330078} {"timestamp_utc": "2026-04-13T09:32:03Z", "mode": "train", "global_step": 862, "epoch": 0.08658965344048217, "loss": 0.1107, "grad_norm": 18.969633102416992, "learning_rate": 7.390909090909092e-06, "num_tokens": 1523613.0, "completions/mean_length": 35.0, "completions/min_length": 26.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.0, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.49755197763442993, "rewards/meter/std": 0.40188562870025635, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9942089319229126, "rewards/repeat_soft/std": 0.007229667156934738, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.17136012017726898, "rewards/total_composite/mean": 0.44823718070983887, "rewards/total_composite/std": 0.14406037330627441, "reward": 0.44823718070983887, "reward_std": 0.14406037330627441, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18669536709785461, "sampling/sampling_logp_difference/max": 2.0277328491210938, "sampling/importance_sampling_ratio/min": 0.23620277643203735, "sampling/importance_sampling_ratio/mean": 0.9968456029891968, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.049236349761486, "clip_ratio/low_mean": 0.0733456676825881, "clip_ratio/low_min": 0.0733456676825881, "clip_ratio/high_mean": 0.09544275049120188, "clip_ratio/high_max": 0.09544275049120188, "clip_ratio/region_mean": 0.16878841817378998, "reward_total_mean": 0.44823718070983887, "reward_meter_mean": 0.49755197763442993, "reward_meter_std": 0.40188562870025635, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9942089319229126, "reward_repeat_soft_std": 0.007229667156934738, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.17136012017726898, "reward_total_composite_mean": 0.44823718070983887, "reward_total_composite_std": 0.14406037330627441} {"timestamp_utc": "2026-04-13T09:32:15Z", "mode": "train", "global_step": 863, "epoch": 0.08669010547463586, "loss": -0.1046, "grad_norm": 2.331528902053833, "learning_rate": 7.3878787878787885e-06, "num_tokens": 1525192.0, "completions/mean_length": 161.375, "completions/min_length": 42.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 44.5, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.665990948677063, "rewards/meter/std": 0.4026106297969818, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9687302112579346, "rewards/repeat_soft/std": 0.023419933393597603, "rewards/judge_quality/mean": 0.33125001192092896, "rewards/judge_quality/std": 0.1738995909690857, "rewards/total_composite/mean": 0.4068011939525604, "rewards/total_composite/std": 0.2627447545528412, "reward": 0.4068011939525604, "reward_std": 0.2627447247505188, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15849895775318146, "sampling/sampling_logp_difference/max": 1.3615283966064453, "sampling/importance_sampling_ratio/min": 0.25626879930496216, "sampling/importance_sampling_ratio/mean": 1.0268831253051758, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9701281487941742, "clip_ratio/low_mean": 0.012019230984151363, "clip_ratio/low_min": 0.012019230984151363, "clip_ratio/high_mean": 0.10455992724746466, "clip_ratio/high_max": 0.10455992724746466, "clip_ratio/region_mean": 0.11657915823161602, "reward_total_mean": 0.4068011939525604, "reward_meter_mean": 0.665990948677063, "reward_meter_std": 0.4026106297969818, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9687302112579346, "reward_repeat_soft_std": 0.023419933393597603, "reward_judge_quality_mean": 0.33125001192092896, "reward_judge_quality_std": 0.1738995909690857, "reward_total_composite_mean": 0.4068011939525604, "reward_total_composite_std": 0.2627447545528412} {"timestamp_utc": "2026-04-13T09:32:26Z", "mode": "train", "global_step": 864, "epoch": 0.08679055750878956, "loss": -0.0993, "grad_norm": 4.113894939422607, "learning_rate": 7.384848484848486e-06, "num_tokens": 1526937.0, "completions/mean_length": 94.125, "completions/min_length": 27.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 34.42857360839844, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.6087803840637207, "rewards/meter/std": 0.2994439899921417, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9791477918624878, "rewards/repeat_soft/std": 0.018508000299334526, "rewards/judge_quality/mean": 0.8149999976158142, "rewards/judge_quality/std": 0.30928489565849304, "rewards/total_composite/mean": 0.6250215768814087, "rewards/total_composite/std": 0.3098732531070709, "reward": 0.6250215768814087, "reward_std": 0.3098732531070709, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1697976291179657, "sampling/sampling_logp_difference/max": 1.9107728004455566, "sampling/importance_sampling_ratio/min": 0.14796599745750427, "sampling/importance_sampling_ratio/mean": 0.9970444440841675, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8175250664353371, "clip_ratio/low_mean": 0.02083333395421505, "clip_ratio/low_min": 0.02083333395421505, "clip_ratio/high_mean": 0.08237126097083092, "clip_ratio/high_max": 0.08237126097083092, "clip_ratio/region_mean": 0.10320459492504597, "reward_total_mean": 0.6250215768814087, "reward_meter_mean": 0.6087803840637207, "reward_meter_std": 0.2994439899921417, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9791477918624878, "reward_repeat_soft_std": 0.018508000299334526, "reward_judge_quality_mean": 0.8149999976158142, "reward_judge_quality_std": 0.30928489565849304, "reward_total_composite_mean": 0.6250215768814087, "reward_total_composite_std": 0.3098732531070709} {"timestamp_utc": "2026-04-13T09:32:37Z", "mode": "train", "global_step": 865, "epoch": 0.08689100954294325, "loss": -0.0878, "grad_norm": 2.10207462310791, "learning_rate": 7.381818181818182e-06, "num_tokens": 1528418.0, "completions/mean_length": 159.125, "completions/min_length": 38.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 41.5, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.613077700138092, "rewards/meter/std": 0.496181458234787, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9650115966796875, "rewards/repeat_soft/std": 0.026163186877965927, "rewards/judge_quality/mean": 0.4012500047683716, "rewards/judge_quality/std": 0.27351874113082886, "rewards/total_composite/mean": 0.4715662896633148, "rewards/total_composite/std": 0.33212050795555115, "reward": 0.4715662896633148, "reward_std": 0.33212053775787354, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11449142545461655, "sampling/sampling_logp_difference/max": 1.433730959892273, "sampling/importance_sampling_ratio/min": 0.23841774463653564, "sampling/importance_sampling_ratio/mean": 1.0337467193603516, "sampling/importance_sampling_ratio/max": 1.96662175655365, "entropy": 0.6000302210450172, "clip_ratio/low_mean": 0.01944444514811039, "clip_ratio/low_min": 0.01944444514811039, "clip_ratio/high_mean": 0.07210694067180157, "clip_ratio/high_max": 0.07210694067180157, "clip_ratio/region_mean": 0.09155138581991196, "reward_total_mean": 0.4715662896633148, "reward_meter_mean": 0.613077700138092, "reward_meter_std": 0.496181458234787, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9650115966796875, "reward_repeat_soft_std": 0.026163186877965927, "reward_judge_quality_mean": 0.4012500047683716, "reward_judge_quality_std": 0.27351874113082886, "reward_total_composite_mean": 0.4715662896633148, "reward_total_composite_std": 0.33212050795555115} {"timestamp_utc": "2026-04-13T09:32:43Z", "mode": "train", "global_step": 866, "epoch": 0.08699146157709693, "loss": -0.0487, "grad_norm": 16.4656925201416, "learning_rate": 7.378787878787879e-06, "num_tokens": 1530311.0, "completions/mean_length": 40.625, "completions/min_length": 31.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.625, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.8736456632614136, "rewards/meter/std": 0.2258608043193817, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.922467827796936, "rewards/repeat_soft/std": 0.048922378569841385, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.6859979033470154, "rewards/total_composite/std": 0.1750345379114151, "reward": 0.6859979033470154, "reward_std": 0.1750345379114151, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18711188435554504, "sampling/sampling_logp_difference/max": 1.5942096710205078, "sampling/importance_sampling_ratio/min": 0.20306895673274994, "sampling/importance_sampling_ratio/mean": 1.008025884628296, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.359382875263691, "clip_ratio/low_mean": 0.0766486581414938, "clip_ratio/low_min": 0.0766486581414938, "clip_ratio/high_mean": 0.06952030956745148, "clip_ratio/high_max": 0.06952030956745148, "clip_ratio/region_mean": 0.14616896770894527, "reward_total_mean": 0.6859979033470154, "reward_meter_mean": 0.8736456632614136, "reward_meter_std": 0.2258608043193817, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.922467827796936, "reward_repeat_soft_std": 0.048922378569841385, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.6859979033470154, "reward_total_composite_std": 0.1750345379114151} {"timestamp_utc": "2026-04-13T09:32:49Z", "mode": "train", "global_step": 867, "epoch": 0.08709191361125063, "loss": -0.0763, "grad_norm": 10.90440559387207, "learning_rate": 7.375757575757576e-06, "num_tokens": 1532103.0, "completions/mean_length": 58.0, "completions/min_length": 47.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.8593844175338745, "rewards/meter/std": 0.3025902509689331, "rewards/count_adherence/mean": 0.7916666865348816, "rewards/count_adherence/std": 0.17251639068126678, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9123940467834473, "rewards/repeat_soft/std": 0.07598268985748291, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.5321553945541382, "rewards/total_composite/std": 0.09145176410675049, "reward": 0.5321553945541382, "reward_std": 0.09145176410675049, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1319979429244995, "sampling/sampling_logp_difference/max": 1.164151668548584, "sampling/importance_sampling_ratio/min": 0.31218740344047546, "sampling/importance_sampling_ratio/mean": 1.0285383462905884, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.027083732187748, "clip_ratio/low_mean": 0.029255319386720657, "clip_ratio/low_min": 0.029255319386720657, "clip_ratio/high_mean": 0.13843377400189638, "clip_ratio/high_max": 0.13843377400189638, "clip_ratio/region_mean": 0.16768909338861704, "reward_total_mean": 0.5321553945541382, "reward_meter_mean": 0.8593844175338745, "reward_meter_std": 0.3025902509689331, "reward_count_adherence_mean": 0.7916666865348816, "reward_count_adherence_std": 0.17251639068126678, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9123940467834473, "reward_repeat_soft_std": 0.07598268985748291, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.5321553945541382, "reward_total_composite_std": 0.09145176410675049} {"timestamp_utc": "2026-04-13T09:33:00Z", "mode": "train", "global_step": 868, "epoch": 0.08719236564540432, "loss": -0.137, "grad_norm": 3.0424578189849854, "learning_rate": 7.372727272727274e-06, "num_tokens": 1533748.0, "completions/mean_length": 107.625, "completions/min_length": 34.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 49.85714340209961, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.8029947280883789, "rewards/meter/std": 0.3323410749435425, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8788893818855286, "rewards/repeat_soft/std": 0.2116212695837021, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.13274572789669037, "rewards/total_composite/mean": 0.4934343099594116, "rewards/total_composite/std": 0.21000738441944122, "reward": 0.4934343099594116, "reward_std": 0.21000738441944122, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16232752799987793, "sampling/sampling_logp_difference/max": 1.2833247184753418, "sampling/importance_sampling_ratio/min": 0.29169297218322754, "sampling/importance_sampling_ratio/mean": 1.0410609245300293, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0614218637347221, "clip_ratio/low_mean": 0.03357380721718073, "clip_ratio/low_min": 0.03357380721718073, "clip_ratio/high_mean": 0.10312587767839432, "clip_ratio/high_max": 0.10312587767839432, "clip_ratio/region_mean": 0.13669968489557505, "reward_total_mean": 0.4934343099594116, "reward_meter_mean": 0.8029947280883789, "reward_meter_std": 0.3323410749435425, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8788893818855286, "reward_repeat_soft_std": 0.2116212695837021, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.13274572789669037, "reward_total_composite_mean": 0.4934343099594116, "reward_total_composite_std": 0.21000738441944122} {"timestamp_utc": "2026-04-13T09:33:06Z", "mode": "train", "global_step": 869, "epoch": 0.08729281767955802, "loss": -0.2605, "grad_norm": 16.398597717285156, "learning_rate": 7.36969696969697e-06, "num_tokens": 1535196.0, "completions/mean_length": 24.0, "completions/min_length": 7.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.0, "completions/min_terminated_length": 7.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.503116250038147, "rewards/meter/std": 0.44779786467552185, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.4629100561141968, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.8853303790092468, "rewards/repeat_soft/std": 0.1647690385580063, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.0843462198972702, "rewards/total_composite/mean": 0.38540714979171753, "rewards/total_composite/std": 0.26252666115760803, "reward": 0.38540714979171753, "reward_std": 0.26252666115760803, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1784038543701172, "sampling/sampling_logp_difference/max": 1.265294075012207, "sampling/importance_sampling_ratio/min": 0.2821563184261322, "sampling/importance_sampling_ratio/mean": 1.0248316526412964, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2060170471668243, "clip_ratio/low_mean": 0.03300865925848484, "clip_ratio/low_min": 0.03300865925848484, "clip_ratio/high_mean": 0.0912698432803154, "clip_ratio/high_max": 0.0912698432803154, "clip_ratio/region_mean": 0.12427850253880024, "reward_total_mean": 0.38540714979171753, "reward_meter_mean": 0.503116250038147, "reward_meter_std": 0.44779786467552185, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.4629100561141968, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.8853303790092468, "reward_repeat_soft_std": 0.1647690385580063, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.0843462198972702, "reward_total_composite_mean": 0.38540714979171753, "reward_total_composite_std": 0.26252666115760803} {"timestamp_utc": "2026-04-13T09:33:18Z", "mode": "train", "global_step": 870, "epoch": 0.0873932697137117, "loss": -0.0392, "grad_norm": 1.135308861732483, "learning_rate": 7.3666666666666676e-06, "num_tokens": 1536690.0, "completions/mean_length": 396.75, "completions/min_length": 49.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.75, "completions/mean_terminated_length": 51.0, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.2545756697654724, "rewards/meter/std": 0.32993805408477783, "rewards/count_adherence/mean": 0.6875, "rewards/count_adherence/std": 0.25877460837364197, "rewards/hard_gate/mean": 0.5, "rewards/hard_gate/std": 0.5345224738121033, "rewards/repeat_soft/mean": 0.9848628044128418, "rewards/repeat_soft/std": 0.016018152236938477, "rewards/judge_quality/mean": 0.14249999821186066, "rewards/judge_quality/std": 0.17127670347690582, "rewards/total_composite/mean": 0.1625368744134903, "rewards/total_composite/std": 0.18057850003242493, "reward": 0.1625368744134903, "reward_std": 0.18057848513126373, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1754874289035797, "sampling/sampling_logp_difference/max": 1.0462274551391602, "sampling/importance_sampling_ratio/min": 0.35126039385795593, "sampling/importance_sampling_ratio/mean": 1.0462087392807007, "sampling/importance_sampling_ratio/max": 1.856386423110962, "entropy": 0.4270182102918625, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.043608009815216064, "clip_ratio/high_max": 0.043608009815216064, "clip_ratio/region_mean": 0.043608009815216064, "reward_total_mean": 0.1625368744134903, "reward_meter_mean": 0.2545756697654724, "reward_meter_std": 0.32993805408477783, "reward_count_adherence_mean": 0.6875, "reward_count_adherence_std": 0.25877460837364197, "reward_hard_gate_mean": 0.5, "reward_hard_gate_std": 0.5345224738121033, "reward_repeat_soft_mean": 0.9848628044128418, "reward_repeat_soft_std": 0.016018152236938477, "reward_judge_quality_mean": 0.14249999821186066, "reward_judge_quality_std": 0.17127670347690582, "reward_total_composite_mean": 0.1625368744134903, "reward_total_composite_std": 0.18057850003242493} {"timestamp_utc": "2026-04-13T09:33:25Z", "mode": "train", "global_step": 871, "epoch": 0.08749372174786539, "loss": 0.0146, "grad_norm": 9.108240127563477, "learning_rate": 7.363636363636364e-06, "num_tokens": 1539043.0, "completions/mean_length": 95.125, "completions/min_length": 77.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.125, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.6485616564750671, "rewards/meter/std": 0.36094608902931213, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8932536840438843, "rewards/repeat_soft/std": 0.05243608355522156, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.43055039644241333, "rewards/total_composite/std": 0.19202134013175964, "reward": 0.43055039644241333, "reward_std": 0.19202131032943726, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16381889581680298, "sampling/sampling_logp_difference/max": 2.3126883506774902, "sampling/importance_sampling_ratio/min": 0.09899476170539856, "sampling/importance_sampling_ratio/mean": 1.0176433324813843, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0679646357893944, "clip_ratio/low_mean": 0.04217723850160837, "clip_ratio/low_min": 0.04217723850160837, "clip_ratio/high_mean": 0.09624836035072803, "clip_ratio/high_max": 0.09624836035072803, "clip_ratio/region_mean": 0.1384255988523364, "reward_total_mean": 0.43055039644241333, "reward_meter_mean": 0.6485616564750671, "reward_meter_std": 0.36094608902931213, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8932536840438843, "reward_repeat_soft_std": 0.05243608355522156, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.43055039644241333, "reward_total_composite_std": 0.19202134013175964} {"timestamp_utc": "2026-04-13T09:33:31Z", "mode": "train", "global_step": 872, "epoch": 0.08759417378201909, "loss": -0.0083, "grad_norm": 17.871868133544922, "learning_rate": 7.360606060606061e-06, "num_tokens": 1540625.0, "completions/mean_length": 36.75, "completions/min_length": 30.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.75, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.327788382768631, "rewards/meter/std": 0.35086458921432495, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9688414335250854, "rewards/repeat_soft/std": 0.0360298790037632, "rewards/judge_quality/mean": 0.5862500071525574, "rewards/judge_quality/std": 0.22984080016613007, "rewards/total_composite/mean": 0.459040105342865, "rewards/total_composite/std": 0.12171979993581772, "reward": 0.459040105342865, "reward_std": 0.12171980738639832, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18134641647338867, "sampling/sampling_logp_difference/max": 1.570383071899414, "sampling/importance_sampling_ratio/min": 0.20796550810337067, "sampling/importance_sampling_ratio/mean": 1.0140125751495361, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.123292826116085, "clip_ratio/low_mean": 0.07075076550245285, "clip_ratio/low_min": 0.07075076550245285, "clip_ratio/high_mean": 0.061073604971170425, "clip_ratio/high_max": 0.061073604971170425, "clip_ratio/region_mean": 0.13182437047362328, "reward_total_mean": 0.459040105342865, "reward_meter_mean": 0.327788382768631, "reward_meter_std": 0.35086458921432495, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9688414335250854, "reward_repeat_soft_std": 0.0360298790037632, "reward_judge_quality_mean": 0.5862500071525574, "reward_judge_quality_std": 0.22984080016613007, "reward_total_composite_mean": 0.459040105342865, "reward_total_composite_std": 0.12171979993581772} {"timestamp_utc": "2026-04-13T09:33:43Z", "mode": "train", "global_step": 873, "epoch": 0.08769462581617278, "loss": -0.1484, "grad_norm": 3.168926477432251, "learning_rate": 7.357575757575758e-06, "num_tokens": 1542634.0, "completions/mean_length": 126.125, "completions/min_length": 61.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 71.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9259101748466492, "rewards/meter/std": 0.0754835233092308, "rewards/count_adherence/mean": 0.6875, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9551552534103394, "rewards/repeat_soft/std": 0.029804719612002373, "rewards/judge_quality/mean": 0.6937500238418579, "rewards/judge_quality/std": 0.3114453852176666, "rewards/total_composite/mean": 0.6748001575469971, "rewards/total_composite/std": 0.29786404967308044, "reward": 0.6748001575469971, "reward_std": 0.29786401987075806, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16060665249824524, "sampling/sampling_logp_difference/max": 1.4686574935913086, "sampling/importance_sampling_ratio/min": 0.23023438453674316, "sampling/importance_sampling_ratio/mean": 1.017259120941162, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0811131298542023, "clip_ratio/low_mean": 0.035211266949772835, "clip_ratio/low_min": 0.035211266949772835, "clip_ratio/high_mean": 0.10306689236313105, "clip_ratio/high_max": 0.10306689236313105, "clip_ratio/region_mean": 0.13827815931290388, "reward_total_mean": 0.6748001575469971, "reward_meter_mean": 0.9259101748466492, "reward_meter_std": 0.0754835233092308, "reward_count_adherence_mean": 0.6875, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9551552534103394, "reward_repeat_soft_std": 0.029804719612002373, "reward_judge_quality_mean": 0.6937500238418579, "reward_judge_quality_std": 0.3114453852176666, "reward_total_composite_mean": 0.6748001575469971, "reward_total_composite_std": 0.29786404967308044} {"timestamp_utc": "2026-04-13T09:33:49Z", "mode": "train", "global_step": 874, "epoch": 0.08779507785032648, "loss": 0.0187, "grad_norm": 12.982820510864258, "learning_rate": 7.354545454545456e-06, "num_tokens": 1544620.0, "completions/mean_length": 65.25, "completions/min_length": 54.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.25, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.5792921781539917, "rewards/meter/std": 0.3442385494709015, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9295393228530884, "rewards/repeat_soft/std": 0.05003296583890915, "rewards/judge_quality/mean": 0.6112500429153442, "rewards/judge_quality/std": 0.20642796158790588, "rewards/total_composite/mean": 0.530864953994751, "rewards/total_composite/std": 0.17147348821163177, "reward": 0.530864953994751, "reward_std": 0.17147347331047058, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16094931960105896, "sampling/sampling_logp_difference/max": 2.14949369430542, "sampling/importance_sampling_ratio/min": 0.11654314398765564, "sampling/importance_sampling_ratio/mean": 1.0058163404464722, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.842586986720562, "clip_ratio/low_mean": 0.053718000650405884, "clip_ratio/low_min": 0.053718000650405884, "clip_ratio/high_mean": 0.08179479092359543, "clip_ratio/high_max": 0.08179479092359543, "clip_ratio/region_mean": 0.1355127915740013, "reward_total_mean": 0.530864953994751, "reward_meter_mean": 0.5792921781539917, "reward_meter_std": 0.3442385494709015, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9295393228530884, "reward_repeat_soft_std": 0.05003296583890915, "reward_judge_quality_mean": 0.6112500429153442, "reward_judge_quality_std": 0.20642796158790588, "reward_total_composite_mean": 0.530864953994751, "reward_total_composite_std": 0.17147348821163177} {"timestamp_utc": "2026-04-13T09:34:04Z", "mode": "train", "global_step": 875, "epoch": 0.08789552988448016, "loss": 0.0378, "grad_norm": 12.344579696655273, "learning_rate": 7.351515151515151e-06, "num_tokens": 1546339.0, "completions/mean_length": 46.875, "completions/min_length": 45.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.875, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.8060281872749329, "rewards/meter/std": 0.2820912301540375, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9610169529914856, "rewards/repeat_soft/std": 0.038200557231903076, "rewards/judge_quality/mean": 0.5487500429153442, "rewards/judge_quality/std": 0.22937415540218353, "rewards/total_composite/mean": 0.6316392421722412, "rewards/total_composite/std": 0.1540820449590683, "reward": 0.6316392421722412, "reward_std": 0.1540820151567459, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15313468873500824, "sampling/sampling_logp_difference/max": 2.062655448913574, "sampling/importance_sampling_ratio/min": 0.12711596488952637, "sampling/importance_sampling_ratio/mean": 1.0250355005264282, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0811816975474358, "clip_ratio/low_mean": 0.09521636413410306, "clip_ratio/low_min": 0.09521636413410306, "clip_ratio/high_mean": 0.02960222028195858, "clip_ratio/high_max": 0.02960222028195858, "clip_ratio/region_mean": 0.12481858441606164, "reward_total_mean": 0.6316392421722412, "reward_meter_mean": 0.8060281872749329, "reward_meter_std": 0.2820912301540375, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9610169529914856, "reward_repeat_soft_std": 0.038200557231903076, "reward_judge_quality_mean": 0.5487500429153442, "reward_judge_quality_std": 0.22937415540218353, "reward_total_composite_mean": 0.6316392421722412, "reward_total_composite_std": 0.1540820449590683} {"timestamp_utc": "2026-04-13T09:34:15Z", "mode": "train", "global_step": 876, "epoch": 0.08799598191863385, "loss": -0.1188, "grad_norm": 2.8642337322235107, "learning_rate": 7.348484848484849e-06, "num_tokens": 1548107.0, "completions/mean_length": 102.0, "completions/min_length": 40.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 43.42857360839844, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.8618472814559937, "rewards/meter/std": 0.13803237676620483, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.87885582447052, "rewards/repeat_soft/std": 0.11330028623342514, "rewards/judge_quality/mean": 0.4300000071525574, "rewards/judge_quality/std": 0.24454039335250854, "rewards/total_composite/mean": 0.5300813913345337, "rewards/total_composite/std": 0.23540335893630981, "reward": 0.5300813913345337, "reward_std": 0.23540334403514862, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12949153780937195, "sampling/sampling_logp_difference/max": 1.6033679246902466, "sampling/importance_sampling_ratio/min": 0.2012176811695099, "sampling/importance_sampling_ratio/mean": 1.0226432085037231, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6162024512887001, "clip_ratio/low_mean": 0.027820121496915817, "clip_ratio/low_min": 0.027820121496915817, "clip_ratio/high_mean": 0.0726192258298397, "clip_ratio/high_max": 0.0726192258298397, "clip_ratio/region_mean": 0.10043934732675552, "reward_total_mean": 0.5300813913345337, "reward_meter_mean": 0.8618472814559937, "reward_meter_std": 0.13803237676620483, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.87885582447052, "reward_repeat_soft_std": 0.11330028623342514, "reward_judge_quality_mean": 0.4300000071525574, "reward_judge_quality_std": 0.24454039335250854, "reward_total_composite_mean": 0.5300813913345337, "reward_total_composite_std": 0.23540335893630981} {"timestamp_utc": "2026-04-13T09:34:26Z", "mode": "train", "global_step": 877, "epoch": 0.08809643395278755, "loss": -0.0845, "grad_norm": 3.6687779426574707, "learning_rate": 7.345454545454546e-06, "num_tokens": 1549526.0, "completions/mean_length": 102.375, "completions/min_length": 38.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 43.85714340209961, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.439591646194458, "rewards/meter/std": 0.34563952684402466, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8043549060821533, "rewards/repeat_soft/std": 0.12364349514245987, "rewards/judge_quality/mean": 0.5649999976158142, "rewards/judge_quality/std": 0.3206244111061096, "rewards/total_composite/mean": 0.47407668828964233, "rewards/total_composite/std": 0.268888384103775, "reward": 0.47407668828964233, "reward_std": 0.268888384103775, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09451233595609665, "sampling/sampling_logp_difference/max": 1.1827497482299805, "sampling/importance_sampling_ratio/min": 0.4081783890724182, "sampling/importance_sampling_ratio/mean": 1.0186411142349243, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5319491885602474, "clip_ratio/low_mean": 0.0352316303178668, "clip_ratio/low_min": 0.0352316303178668, "clip_ratio/high_mean": 0.04231360787525773, "clip_ratio/high_max": 0.04231360787525773, "clip_ratio/region_mean": 0.07754523819312453, "reward_total_mean": 0.47407668828964233, "reward_meter_mean": 0.439591646194458, "reward_meter_std": 0.34563952684402466, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8043549060821533, "reward_repeat_soft_std": 0.12364349514245987, "reward_judge_quality_mean": 0.5649999976158142, "reward_judge_quality_std": 0.3206244111061096, "reward_total_composite_mean": 0.47407668828964233, "reward_total_composite_std": 0.268888384103775} {"timestamp_utc": "2026-04-13T09:34:35Z", "mode": "train", "global_step": 878, "epoch": 0.08819688598694124, "loss": 0.888, "grad_norm": 15.310219764709473, "learning_rate": 7.342424242424243e-06, "num_tokens": 1551335.0, "completions/mean_length": 59.125, "completions/min_length": 33.0, "completions/max_length": 219.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.125, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 219.0, "rewards/meter/mean": 0.3386475443840027, "rewards/meter/std": 0.3047406077384949, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8161276578903198, "rewards/repeat_soft/std": 0.26778846979141235, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.0843462198972702, "rewards/total_composite/mean": 0.3847002387046814, "rewards/total_composite/std": 0.1626531183719635, "reward": 0.3847002387046814, "reward_std": 0.1626531183719635, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12099866569042206, "sampling/sampling_logp_difference/max": 2.6160330772399902, "sampling/importance_sampling_ratio/min": 0.07309223711490631, "sampling/importance_sampling_ratio/mean": 1.0092995166778564, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0028022192418575, "clip_ratio/low_mean": 0.02445640368387103, "clip_ratio/low_min": 0.02445640368387103, "clip_ratio/high_mean": 0.11491053458303213, "clip_ratio/high_max": 0.11491053458303213, "clip_ratio/region_mean": 0.13936693826690316, "reward_total_mean": 0.3847002387046814, "reward_meter_mean": 0.3386475443840027, "reward_meter_std": 0.3047406077384949, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8161276578903198, "reward_repeat_soft_std": 0.26778846979141235, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.0843462198972702, "reward_total_composite_mean": 0.3847002387046814, "reward_total_composite_std": 0.1626531183719635} {"timestamp_utc": "2026-04-13T09:34:41Z", "mode": "train", "global_step": 879, "epoch": 0.08829733802109492, "loss": 0.0035, "grad_norm": 15.814071655273438, "learning_rate": 7.3393939393939395e-06, "num_tokens": 1552783.0, "completions/mean_length": 39.0, "completions/min_length": 31.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.0, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.34972092509269714, "rewards/meter/std": 0.42591872811317444, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9957960844039917, "rewards/repeat_soft/std": 0.006417128257453442, "rewards/judge_quality/mean": 0.6812499761581421, "rewards/judge_quality/std": 0.2231871634721756, "rewards/total_composite/mean": 0.48811185359954834, "rewards/total_composite/std": 0.2064460664987564, "reward": 0.48811185359954834, "reward_std": 0.2064460664987564, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1526462584733963, "sampling/sampling_logp_difference/max": 2.4089221954345703, "sampling/importance_sampling_ratio/min": 0.2115991711616516, "sampling/importance_sampling_ratio/mean": 1.0094772577285767, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7610408887267113, "clip_ratio/low_mean": 0.1086641252040863, "clip_ratio/low_min": 0.1086641252040863, "clip_ratio/high_mean": 0.054252199828624725, "clip_ratio/high_max": 0.054252199828624725, "clip_ratio/region_mean": 0.16291632503271103, "reward_total_mean": 0.48811185359954834, "reward_meter_mean": 0.34972092509269714, "reward_meter_std": 0.42591872811317444, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9957960844039917, "reward_repeat_soft_std": 0.006417128257453442, "reward_judge_quality_mean": 0.6812499761581421, "reward_judge_quality_std": 0.2231871634721756, "reward_total_composite_mean": 0.48811185359954834, "reward_total_composite_std": 0.2064460664987564} {"timestamp_utc": "2026-04-13T09:34:47Z", "mode": "train", "global_step": 880, "epoch": 0.08839779005524862, "loss": 0.0044, "grad_norm": 10.868603706359863, "learning_rate": 7.336363636363637e-06, "num_tokens": 1554682.0, "completions/mean_length": 72.375, "completions/min_length": 64.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.375, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.9660510420799255, "rewards/meter/std": 0.03841925039887428, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8535215258598328, "rewards/repeat_soft/std": 0.12191952764987946, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.631879448890686, "rewards/total_composite/std": 0.12384099513292313, "reward": 0.631879448890686, "reward_std": 0.12384098768234253, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15149742364883423, "sampling/sampling_logp_difference/max": 2.774995803833008, "sampling/importance_sampling_ratio/min": 0.06234974041581154, "sampling/importance_sampling_ratio/mean": 1.021327018737793, "sampling/importance_sampling_ratio/max": 1.9698742628097534, "entropy": 1.0442497804760933, "clip_ratio/low_mean": 0.09328743256628513, "clip_ratio/low_min": 0.09328743256628513, "clip_ratio/high_mean": 0.01666666753590107, "clip_ratio/high_max": 0.01666666753590107, "clip_ratio/region_mean": 0.1099541001021862, "reward_total_mean": 0.631879448890686, "reward_meter_mean": 0.9660510420799255, "reward_meter_std": 0.03841925039887428, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8535215258598328, "reward_repeat_soft_std": 0.12191952764987946, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.631879448890686, "reward_total_composite_std": 0.12384099513292313} {"timestamp_utc": "2026-04-13T09:34:55Z", "mode": "train", "global_step": 881, "epoch": 0.08849824208940231, "loss": -0.0457, "grad_norm": 8.910069465637207, "learning_rate": 7.333333333333333e-06, "num_tokens": 1557133.0, "completions/mean_length": 78.375, "completions/min_length": 66.0, "completions/max_length": 100.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.375, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 100.0, "rewards/meter/mean": 0.9773383140563965, "rewards/meter/std": 0.019956661388278008, "rewards/count_adherence/mean": 0.84375, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8784147500991821, "rewards/repeat_soft/std": 0.08408968150615692, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5673255920410156, "rewards/total_composite/std": 0.019805608317255974, "reward": 0.5673255920410156, "reward_std": 0.019805599004030228, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1358868032693863, "sampling/sampling_logp_difference/max": 1.8858757019042969, "sampling/importance_sampling_ratio/min": 0.253724604845047, "sampling/importance_sampling_ratio/mean": 1.027084469795227, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1675440669059753, "clip_ratio/low_mean": 0.06719556078314781, "clip_ratio/low_min": 0.06719556078314781, "clip_ratio/high_mean": 0.055186279118061066, "clip_ratio/high_max": 0.055186279118061066, "clip_ratio/region_mean": 0.12238183990120888, "reward_total_mean": 0.5673255920410156, "reward_meter_mean": 0.9773383140563965, "reward_meter_std": 0.019956661388278008, "reward_count_adherence_mean": 0.84375, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8784147500991821, "reward_repeat_soft_std": 0.08408968150615692, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5673255920410156, "reward_total_composite_std": 0.019805608317255974} {"timestamp_utc": "2026-04-13T09:35:02Z", "mode": "train", "global_step": 882, "epoch": 0.088598694123556, "loss": 0.0344, "grad_norm": 9.84181022644043, "learning_rate": 7.330303030303031e-06, "num_tokens": 1559327.0, "completions/mean_length": 93.25, "completions/min_length": 76.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.25, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.9091979265213013, "rewards/meter/std": 0.159868061542511, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8407905101776123, "rewards/repeat_soft/std": 0.06750401109457016, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.5818697214126587, "rewards/total_composite/std": 0.10077501088380814, "reward": 0.5818697214126587, "reward_std": 0.10077502578496933, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12708139419555664, "sampling/sampling_logp_difference/max": 3.6555018424987793, "sampling/importance_sampling_ratio/min": 0.025848522782325745, "sampling/importance_sampling_ratio/mean": 1.0060118436813354, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8384061381220818, "clip_ratio/low_mean": 0.07911044172942638, "clip_ratio/low_min": 0.07911044172942638, "clip_ratio/high_mean": 0.029835755936801434, "clip_ratio/high_max": 0.029835755936801434, "clip_ratio/region_mean": 0.10894619766622782, "reward_total_mean": 0.5818697214126587, "reward_meter_mean": 0.9091979265213013, "reward_meter_std": 0.159868061542511, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8407905101776123, "reward_repeat_soft_std": 0.06750401109457016, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.5818697214126587, "reward_total_composite_std": 0.10077501088380814} {"timestamp_utc": "2026-04-13T09:35:09Z", "mode": "train", "global_step": 883, "epoch": 0.0886991461577097, "loss": 0.0462, "grad_norm": 28.119281768798828, "learning_rate": 7.3272727272727285e-06, "num_tokens": 1561064.0, "completions/mean_length": 45.125, "completions/min_length": 41.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.125, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.4719432592391968, "rewards/meter/std": 0.4077528417110443, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.984849750995636, "rewards/repeat_soft/std": 0.013406476937234402, "rewards/judge_quality/mean": 0.48624998331069946, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.4829104244709015, "rewards/total_composite/std": 0.11041922122240067, "reward": 0.4829104244709015, "reward_std": 0.11041922122240067, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18426677584648132, "sampling/sampling_logp_difference/max": 1.7394871711730957, "sampling/importance_sampling_ratio/min": 0.17561043798923492, "sampling/importance_sampling_ratio/mean": 1.00497305393219, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1275340244174004, "clip_ratio/low_mean": 0.07357970625162125, "clip_ratio/low_min": 0.07357970625162125, "clip_ratio/high_mean": 0.0731347594410181, "clip_ratio/high_max": 0.0731347594410181, "clip_ratio/region_mean": 0.14671446569263935, "reward_total_mean": 0.4829104244709015, "reward_meter_mean": 0.4719432592391968, "reward_meter_std": 0.4077528417110443, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.984849750995636, "reward_repeat_soft_std": 0.013406476937234402, "reward_judge_quality_mean": 0.48624998331069946, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.4829104244709015, "reward_total_composite_std": 0.11041922122240067} {"timestamp_utc": "2026-04-13T09:35:20Z", "mode": "train", "global_step": 884, "epoch": 0.08879959819186338, "loss": -0.1798, "grad_norm": 3.204331874847412, "learning_rate": 7.324242424242425e-06, "num_tokens": 1563447.0, "completions/mean_length": 160.875, "completions/min_length": 91.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 110.71429443359375, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.6648643612861633, "rewards/meter/std": 0.432773232460022, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.263523131608963, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.7227896451950073, "rewards/repeat_soft/std": 0.21885265409946442, "rewards/judge_quality/mean": 0.3399999737739563, "rewards/judge_quality/std": 0.15052290260791779, "rewards/total_composite/mean": 0.3958885669708252, "rewards/total_composite/std": 0.2000521868467331, "reward": 0.3958885669708252, "reward_std": 0.2000521868467331, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12298968434333801, "sampling/sampling_logp_difference/max": 1.3381690979003906, "sampling/importance_sampling_ratio/min": 0.2623255252838135, "sampling/importance_sampling_ratio/mean": 1.0034453868865967, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6497861333191395, "clip_ratio/low_mean": 0.0012886597542092204, "clip_ratio/low_min": 0.0012886597542092204, "clip_ratio/high_mean": 0.10129512473940849, "clip_ratio/high_max": 0.10129512473940849, "clip_ratio/region_mean": 0.10258378449361771, "reward_total_mean": 0.3958885669708252, "reward_meter_mean": 0.6648643612861633, "reward_meter_std": 0.432773232460022, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.263523131608963, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.7227896451950073, "reward_repeat_soft_std": 0.21885265409946442, "reward_judge_quality_mean": 0.3399999737739563, "reward_judge_quality_std": 0.15052290260791779, "reward_total_composite_mean": 0.3958885669708252, "reward_total_composite_std": 0.2000521868467331} {"timestamp_utc": "2026-04-13T09:35:31Z", "mode": "train", "global_step": 885, "epoch": 0.08890005022601707, "loss": -0.093, "grad_norm": 5.706384658813477, "learning_rate": 7.321212121212122e-06, "num_tokens": 1565490.0, "completions/mean_length": 141.375, "completions/min_length": 78.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 88.42857360839844, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 110.0, "rewards/meter/mean": 0.7723007202148438, "rewards/meter/std": 0.30487778782844543, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9757391214370728, "rewards/repeat_soft/std": 0.018861595541238785, "rewards/judge_quality/mean": 0.5612499713897705, "rewards/judge_quality/std": 0.3223324716091156, "rewards/total_composite/mean": 0.532701849937439, "rewards/total_composite/std": 0.344864159822464, "reward": 0.532701849937439, "reward_std": 0.344864159822464, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1807408481836319, "sampling/sampling_logp_difference/max": 2.2748515605926514, "sampling/importance_sampling_ratio/min": 0.10281217098236084, "sampling/importance_sampling_ratio/mean": 1.0091204643249512, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0115512311458588, "clip_ratio/low_mean": 0.01692708395421505, "clip_ratio/low_min": 0.01692708395421505, "clip_ratio/high_mean": 0.15578365325927734, "clip_ratio/high_max": 0.15578365325927734, "clip_ratio/region_mean": 0.1727107372134924, "reward_total_mean": 0.532701849937439, "reward_meter_mean": 0.7723007202148438, "reward_meter_std": 0.30487778782844543, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9757391214370728, "reward_repeat_soft_std": 0.018861595541238785, "reward_judge_quality_mean": 0.5612499713897705, "reward_judge_quality_std": 0.3223324716091156, "reward_total_composite_mean": 0.532701849937439, "reward_total_composite_std": 0.344864159822464} {"timestamp_utc": "2026-04-13T09:35:38Z", "mode": "train", "global_step": 886, "epoch": 0.08900050226017077, "loss": 0.1349, "grad_norm": 22.54247283935547, "learning_rate": 7.3181818181818186e-06, "num_tokens": 1567018.0, "completions/mean_length": 24.0, "completions/min_length": 20.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.0, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.7209341526031494, "rewards/meter/std": 0.404031366109848, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9467312693595886, "rewards/repeat_soft/std": 0.03042098507285118, "rewards/judge_quality/mean": 0.45249998569488525, "rewards/judge_quality/std": 0.21224987506866455, "rewards/total_composite/mean": 0.5589122772216797, "rewards/total_composite/std": 0.18516050279140472, "reward": 0.5589122772216797, "reward_std": 0.18516047298908234, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19285623729228973, "sampling/sampling_logp_difference/max": 1.099189043045044, "sampling/importance_sampling_ratio/min": 0.33314111828804016, "sampling/importance_sampling_ratio/mean": 1.0386865139007568, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4280219674110413, "clip_ratio/low_mean": 0.07509615365415812, "clip_ratio/low_min": 0.07509615365415812, "clip_ratio/high_mean": 0.09872482949867845, "clip_ratio/high_max": 0.09872482949867845, "clip_ratio/region_mean": 0.17382098315283656, "reward_total_mean": 0.5589122772216797, "reward_meter_mean": 0.7209341526031494, "reward_meter_std": 0.404031366109848, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9467312693595886, "reward_repeat_soft_std": 0.03042098507285118, "reward_judge_quality_mean": 0.45249998569488525, "reward_judge_quality_std": 0.21224987506866455, "reward_total_composite_mean": 0.5589122772216797, "reward_total_composite_std": 0.18516050279140472} {"timestamp_utc": "2026-04-13T09:35:44Z", "mode": "train", "global_step": 887, "epoch": 0.08910095429432446, "loss": -0.0112, "grad_norm": 22.737993240356445, "learning_rate": 7.315151515151516e-06, "num_tokens": 1568391.0, "completions/mean_length": 27.625, "completions/min_length": 22.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.625, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.5176418423652649, "rewards/meter/std": 0.4330809414386749, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9610613584518433, "rewards/repeat_soft/std": 0.004068940877914429, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.522869348526001, "rewards/total_composite/std": 0.17281579971313477, "reward": 0.522869348526001, "reward_std": 0.17281579971313477, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20167115330696106, "sampling/sampling_logp_difference/max": 1.5116100311279297, "sampling/importance_sampling_ratio/min": 0.22055459022521973, "sampling/importance_sampling_ratio/mean": 1.0319517850875854, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.278345376253128, "clip_ratio/low_mean": 0.07225524540990591, "clip_ratio/low_min": 0.07225524540990591, "clip_ratio/high_mean": 0.11453500390052795, "clip_ratio/high_max": 0.11453500390052795, "clip_ratio/region_mean": 0.18679024931043386, "reward_total_mean": 0.522869348526001, "reward_meter_mean": 0.5176418423652649, "reward_meter_std": 0.4330809414386749, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9610613584518433, "reward_repeat_soft_std": 0.004068940877914429, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.522869348526001, "reward_total_composite_std": 0.17281579971313477} {"timestamp_utc": "2026-04-13T09:35:51Z", "mode": "train", "global_step": 888, "epoch": 0.08920140632847816, "loss": 0.0598, "grad_norm": 10.03502082824707, "learning_rate": 7.312121212121212e-06, "num_tokens": 1570540.0, "completions/mean_length": 96.625, "completions/min_length": 87.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.625, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.9755557179450989, "rewards/meter/std": 0.02128848060965538, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9463392496109009, "rewards/repeat_soft/std": 0.04638994112610817, "rewards/judge_quality/mean": 0.6575000286102295, "rewards/judge_quality/std": 0.1505940854549408, "rewards/total_composite/mean": 0.758562445640564, "rewards/total_composite/std": 0.09291400015354156, "reward": 0.758562445640564, "reward_std": 0.09291400015354156, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13130463659763336, "sampling/sampling_logp_difference/max": 2.592618942260742, "sampling/importance_sampling_ratio/min": 0.07482381910085678, "sampling/importance_sampling_ratio/mean": 1.0128514766693115, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.900644063949585, "clip_ratio/low_mean": 0.021150990389287472, "clip_ratio/low_min": 0.021150990389287472, "clip_ratio/high_mean": 0.09297308884561062, "clip_ratio/high_max": 0.09297308884561062, "clip_ratio/region_mean": 0.11412407923489809, "reward_total_mean": 0.758562445640564, "reward_meter_mean": 0.9755557179450989, "reward_meter_std": 0.02128848060965538, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9463392496109009, "reward_repeat_soft_std": 0.04638994112610817, "reward_judge_quality_mean": 0.6575000286102295, "reward_judge_quality_std": 0.1505940854549408, "reward_total_composite_mean": 0.758562445640564, "reward_total_composite_std": 0.09291400015354156} {"timestamp_utc": "2026-04-13T09:35:58Z", "mode": "train", "global_step": 889, "epoch": 0.08930185836263184, "loss": -0.0072, "grad_norm": 14.553848266601562, "learning_rate": 7.30909090909091e-06, "num_tokens": 1572149.0, "completions/mean_length": 48.125, "completions/min_length": 39.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.125, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.31065088510513306, "rewards/meter/std": 0.2979934513568878, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9635153412818909, "rewards/repeat_soft/std": 0.04018327593803406, "rewards/judge_quality/mean": 0.6237499713897705, "rewards/judge_quality/std": 0.22890658676624298, "rewards/total_composite/mean": 0.465585321187973, "rewards/total_composite/std": 0.10353792458772659, "reward": 0.465585321187973, "reward_std": 0.1035379096865654, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11885638535022736, "sampling/sampling_logp_difference/max": 1.5418891906738281, "sampling/importance_sampling_ratio/min": 0.21397648751735687, "sampling/importance_sampling_ratio/mean": 1.0027971267700195, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5610169097781181, "clip_ratio/low_mean": 0.0477488050237298, "clip_ratio/low_min": 0.0477488050237298, "clip_ratio/high_mean": 0.0557402353733778, "clip_ratio/high_max": 0.0557402353733778, "clip_ratio/region_mean": 0.1034890403971076, "reward_total_mean": 0.465585321187973, "reward_meter_mean": 0.31065088510513306, "reward_meter_std": 0.2979934513568878, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9635153412818909, "reward_repeat_soft_std": 0.04018327593803406, "reward_judge_quality_mean": 0.6237499713897705, "reward_judge_quality_std": 0.22890658676624298, "reward_total_composite_mean": 0.465585321187973, "reward_total_composite_std": 0.10353792458772659} {"timestamp_utc": "2026-04-13T09:36:04Z", "mode": "train", "global_step": 890, "epoch": 0.08940231039678553, "loss": -0.0118, "grad_norm": 11.42885971069336, "learning_rate": 7.306060606060607e-06, "num_tokens": 1573790.0, "completions/mean_length": 55.125, "completions/min_length": 50.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.125, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.672086238861084, "rewards/meter/std": 0.4170806109905243, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6968518495559692, "rewards/repeat_soft/std": 0.13996635377407074, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.49258965253829956, "rewards/total_composite/std": 0.10857101529836655, "reward": 0.49258965253829956, "reward_std": 0.10857101529836655, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13281306624412537, "sampling/sampling_logp_difference/max": 1.5750236511230469, "sampling/importance_sampling_ratio/min": 0.207002654671669, "sampling/importance_sampling_ratio/mean": 1.0106837749481201, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9696046486496925, "clip_ratio/low_mean": 0.051568303257226944, "clip_ratio/low_min": 0.051568303257226944, "clip_ratio/high_mean": 0.07031287346035242, "clip_ratio/high_max": 0.07031287346035242, "clip_ratio/region_mean": 0.12188117671757936, "reward_total_mean": 0.49258965253829956, "reward_meter_mean": 0.672086238861084, "reward_meter_std": 0.4170806109905243, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6968518495559692, "reward_repeat_soft_std": 0.13996635377407074, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.49258965253829956, "reward_total_composite_std": 0.10857101529836655} {"timestamp_utc": "2026-04-13T09:36:10Z", "mode": "train", "global_step": 891, "epoch": 0.08950276243093923, "loss": -0.0165, "grad_norm": 19.228656768798828, "learning_rate": 7.303030303030304e-06, "num_tokens": 1575320.0, "completions/mean_length": 39.25, "completions/min_length": 32.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.25, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.8273656368255615, "rewards/meter/std": 0.2690790593624115, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9214805364608765, "rewards/repeat_soft/std": 0.09300963580608368, "rewards/judge_quality/mean": 0.5900000333786011, "rewards/judge_quality/std": 0.27994900941848755, "rewards/total_composite/mean": 0.6273754239082336, "rewards/total_composite/std": 0.28964903950691223, "reward": 0.6273754239082336, "reward_std": 0.28964903950691223, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15418635308742523, "sampling/sampling_logp_difference/max": 2.8024725914001465, "sampling/importance_sampling_ratio/min": 0.06065988913178444, "sampling/importance_sampling_ratio/mean": 1.008238673210144, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6868798024952412, "clip_ratio/low_mean": 0.10877399798482656, "clip_ratio/low_min": 0.10877399798482656, "clip_ratio/high_mean": 0.05127991968765855, "clip_ratio/high_max": 0.05127991968765855, "clip_ratio/region_mean": 0.1600539176724851, "reward_total_mean": 0.6273754239082336, "reward_meter_mean": 0.8273656368255615, "reward_meter_std": 0.2690790593624115, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9214805364608765, "reward_repeat_soft_std": 0.09300963580608368, "reward_judge_quality_mean": 0.5900000333786011, "reward_judge_quality_std": 0.27994900941848755, "reward_total_composite_mean": 0.6273754239082336, "reward_total_composite_std": 0.28964903950691223} {"timestamp_utc": "2026-04-13T09:36:17Z", "mode": "train", "global_step": 892, "epoch": 0.08960321446509292, "loss": 0.0577, "grad_norm": 13.683673858642578, "learning_rate": 7.3e-06, "num_tokens": 1577103.0, "completions/mean_length": 47.875, "completions/min_length": 40.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.875, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.6726229786872864, "rewards/meter/std": 0.36972475051879883, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9640060663223267, "rewards/repeat_soft/std": 0.04401865229010582, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.17136012017726898, "rewards/total_composite/mean": 0.5704629421234131, "rewards/total_composite/std": 0.1644839495420456, "reward": 0.5704629421234131, "reward_std": 0.1644839495420456, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17061322927474976, "sampling/sampling_logp_difference/max": 1.981736183166504, "sampling/importance_sampling_ratio/min": 0.1378297358751297, "sampling/importance_sampling_ratio/mean": 0.9951289892196655, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1203052997589111, "clip_ratio/low_mean": 0.09358753263950348, "clip_ratio/low_min": 0.09358753263950348, "clip_ratio/high_mean": 0.09015802852809429, "clip_ratio/high_max": 0.09015802852809429, "clip_ratio/region_mean": 0.18374556116759777, "reward_total_mean": 0.5704629421234131, "reward_meter_mean": 0.6726229786872864, "reward_meter_std": 0.36972475051879883, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9640060663223267, "reward_repeat_soft_std": 0.04401865229010582, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.17136012017726898, "reward_total_composite_mean": 0.5704629421234131, "reward_total_composite_std": 0.1644839495420456} {"timestamp_utc": "2026-04-13T09:36:23Z", "mode": "train", "global_step": 893, "epoch": 0.0897036664992466, "loss": 0.0362, "grad_norm": 21.54730224609375, "learning_rate": 7.296969696969698e-06, "num_tokens": 1578534.0, "completions/mean_length": 31.875, "completions/min_length": 28.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.875, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.6040737628936768, "rewards/meter/std": 0.4211108982563019, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9561255574226379, "rewards/repeat_soft/std": 0.028681941330432892, "rewards/judge_quality/mean": 0.45249998569488525, "rewards/judge_quality/std": 0.06902381032705307, "rewards/total_composite/mean": 0.524419903755188, "rewards/total_composite/std": 0.12749241292476654, "reward": 0.524419903755188, "reward_std": 0.12749241292476654, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1891103833913803, "sampling/sampling_logp_difference/max": 1.9380428791046143, "sampling/importance_sampling_ratio/min": 0.14398548007011414, "sampling/importance_sampling_ratio/mean": 1.0233358144760132, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1149725764989853, "clip_ratio/low_mean": 0.04424442304298282, "clip_ratio/low_min": 0.04424442304298282, "clip_ratio/high_mean": 0.10814394149929285, "clip_ratio/high_max": 0.10814394149929285, "clip_ratio/region_mean": 0.15238836454227567, "reward_total_mean": 0.524419903755188, "reward_meter_mean": 0.6040737628936768, "reward_meter_std": 0.4211108982563019, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9561255574226379, "reward_repeat_soft_std": 0.028681941330432892, "reward_judge_quality_mean": 0.45249998569488525, "reward_judge_quality_std": 0.06902381032705307, "reward_total_composite_mean": 0.524419903755188, "reward_total_composite_std": 0.12749241292476654} {"timestamp_utc": "2026-04-13T09:36:29Z", "mode": "train", "global_step": 894, "epoch": 0.0898041185334003, "loss": 0.0664, "grad_norm": 18.897003173828125, "learning_rate": 7.293939393939394e-06, "num_tokens": 1580012.0, "completions/mean_length": 25.75, "completions/min_length": 19.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.75, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.660660982131958, "rewards/meter/std": 0.4115402400493622, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9652037620544434, "rewards/repeat_soft/std": 0.007647482678294182, "rewards/judge_quality/mean": 0.6150000095367432, "rewards/judge_quality/std": 0.25286927819252014, "rewards/total_composite/mean": 0.5076031684875488, "rewards/total_composite/std": 0.3156152367591858, "reward": 0.5076031684875488, "reward_std": 0.3156152069568634, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18812528252601624, "sampling/sampling_logp_difference/max": 1.7578144073486328, "sampling/importance_sampling_ratio/min": 0.17242130637168884, "sampling/importance_sampling_ratio/mean": 0.9939836263656616, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1995478235185146, "clip_ratio/low_mean": 0.0933492835611105, "clip_ratio/low_min": 0.0933492835611105, "clip_ratio/high_mean": 0.11678571626543999, "clip_ratio/high_max": 0.11678571626543999, "clip_ratio/region_mean": 0.21013499982655048, "reward_total_mean": 0.5076031684875488, "reward_meter_mean": 0.660660982131958, "reward_meter_std": 0.4115402400493622, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9652037620544434, "reward_repeat_soft_std": 0.007647482678294182, "reward_judge_quality_mean": 0.6150000095367432, "reward_judge_quality_std": 0.25286927819252014, "reward_total_composite_mean": 0.5076031684875488, "reward_total_composite_std": 0.3156152367591858} {"timestamp_utc": "2026-04-13T09:36:35Z", "mode": "train", "global_step": 895, "epoch": 0.08990457056755399, "loss": 0.0457, "grad_norm": 19.292024612426758, "learning_rate": 7.290909090909092e-06, "num_tokens": 1581940.0, "completions/mean_length": 72.0, "completions/min_length": 58.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.0, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.280703067779541, "rewards/meter/std": 0.2876078188419342, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9423829913139343, "rewards/repeat_soft/std": 0.025699563324451447, "rewards/judge_quality/mean": 0.5612499713897705, "rewards/judge_quality/std": 0.25614938139915466, "rewards/total_composite/mean": 0.4261462688446045, "rewards/total_composite/std": 0.24164758622646332, "reward": 0.4261462688446045, "reward_std": 0.24164758622646332, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.158335879445076, "sampling/sampling_logp_difference/max": 2.8191614151000977, "sampling/importance_sampling_ratio/min": 0.05965594947338104, "sampling/importance_sampling_ratio/mean": 1.0279756784439087, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9015419781208038, "clip_ratio/low_mean": 0.07797630783170462, "clip_ratio/low_min": 0.07797630783170462, "clip_ratio/high_mean": 0.05729399900883436, "clip_ratio/high_max": 0.05729399900883436, "clip_ratio/region_mean": 0.13527030684053898, "reward_total_mean": 0.4261462688446045, "reward_meter_mean": 0.280703067779541, "reward_meter_std": 0.2876078188419342, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9423829913139343, "reward_repeat_soft_std": 0.025699563324451447, "reward_judge_quality_mean": 0.5612499713897705, "reward_judge_quality_std": 0.25614938139915466, "reward_total_composite_mean": 0.4261462688446045, "reward_total_composite_std": 0.24164758622646332} {"timestamp_utc": "2026-04-13T09:36:41Z", "mode": "train", "global_step": 896, "epoch": 0.09000502260170769, "loss": 0.1358, "grad_norm": 20.5400447845459, "learning_rate": 7.287878787878789e-06, "num_tokens": 1583598.0, "completions/mean_length": 36.25, "completions/min_length": 29.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.25, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.6870586276054382, "rewards/meter/std": 0.300336092710495, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9921655058860779, "rewards/repeat_soft/std": 0.010928143747150898, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.22720351815223694, "rewards/total_composite/mean": 0.6126821041107178, "rewards/total_composite/std": 0.18589404225349426, "reward": 0.6126821041107178, "reward_std": 0.18589404225349426, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18025875091552734, "sampling/sampling_logp_difference/max": 2.0302717685699463, "sampling/importance_sampling_ratio/min": 0.13129983842372894, "sampling/importance_sampling_ratio/mean": 1.0002015829086304, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.904352992773056, "clip_ratio/low_mean": 0.10969074815511703, "clip_ratio/low_min": 0.10969074815511703, "clip_ratio/high_mean": 0.031870429404079914, "clip_ratio/high_max": 0.031870429404079914, "clip_ratio/region_mean": 0.14156117755919695, "reward_total_mean": 0.6126821041107178, "reward_meter_mean": 0.6870586276054382, "reward_meter_std": 0.300336092710495, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9921655058860779, "reward_repeat_soft_std": 0.010928143747150898, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.22720351815223694, "reward_total_composite_mean": 0.6126821041107178, "reward_total_composite_std": 0.18589404225349426} {"timestamp_utc": "2026-04-13T09:36:48Z", "mode": "train", "global_step": 897, "epoch": 0.09010547463586138, "loss": 0.0058, "grad_norm": 13.415826797485352, "learning_rate": 7.284848484848486e-06, "num_tokens": 1585191.0, "completions/mean_length": 50.125, "completions/min_length": 42.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.125, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.8849648237228394, "rewards/meter/std": 0.11630762368440628, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9979231953620911, "rewards/repeat_soft/std": 0.004281661007553339, "rewards/judge_quality/mean": 0.7775000333786011, "rewards/judge_quality/std": 0.21359175443649292, "rewards/total_composite/mean": 0.7969489693641663, "rewards/total_composite/std": 0.1330748051404953, "reward": 0.7969489693641663, "reward_std": 0.1330747753381729, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1257050335407257, "sampling/sampling_logp_difference/max": 1.7089028358459473, "sampling/importance_sampling_ratio/min": 0.18106433749198914, "sampling/importance_sampling_ratio/mean": 1.0011931657791138, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8672522008419037, "clip_ratio/low_mean": 0.02392399311065674, "clip_ratio/low_min": 0.02392399311065674, "clip_ratio/high_mean": 0.09560730308294296, "clip_ratio/high_max": 0.09560730308294296, "clip_ratio/region_mean": 0.1195312961935997, "reward_total_mean": 0.7969489693641663, "reward_meter_mean": 0.8849648237228394, "reward_meter_std": 0.11630762368440628, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9979231953620911, "reward_repeat_soft_std": 0.004281661007553339, "reward_judge_quality_mean": 0.7775000333786011, "reward_judge_quality_std": 0.21359175443649292, "reward_total_composite_mean": 0.7969489693641663, "reward_total_composite_std": 0.1330748051404953} {"timestamp_utc": "2026-04-13T09:36:59Z", "mode": "train", "global_step": 898, "epoch": 0.09020592667001506, "loss": -0.06, "grad_norm": 2.393796920776367, "learning_rate": 7.281818181818182e-06, "num_tokens": 1586526.0, "completions/mean_length": 148.875, "completions/min_length": 26.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 27.83333396911621, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.44433021545410156, "rewards/meter/std": 0.4191552996635437, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.4629100561141968, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.8499475717544556, "rewards/repeat_soft/std": 0.3267112970352173, "rewards/judge_quality/mean": 0.30124998092651367, "rewards/judge_quality/std": 0.18349288403987885, "rewards/total_composite/mean": 0.3642628490924835, "rewards/total_composite/std": 0.2603684365749359, "reward": 0.3642628490924835, "reward_std": 0.2603684067726135, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1771446168422699, "sampling/sampling_logp_difference/max": 1.3258414268493652, "sampling/importance_sampling_ratio/min": 0.3767942190170288, "sampling/importance_sampling_ratio/mean": 1.0350534915924072, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8279794976115227, "clip_ratio/low_mean": 0.024193547666072845, "clip_ratio/low_min": 0.024193547666072845, "clip_ratio/high_mean": 0.10722934734076262, "clip_ratio/high_max": 0.10722934734076262, "clip_ratio/region_mean": 0.13142289500683546, "reward_total_mean": 0.3642628490924835, "reward_meter_mean": 0.44433021545410156, "reward_meter_std": 0.4191552996635437, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.4629100561141968, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.8499475717544556, "reward_repeat_soft_std": 0.3267112970352173, "reward_judge_quality_mean": 0.30124998092651367, "reward_judge_quality_std": 0.18349288403987885, "reward_total_composite_mean": 0.3642628490924835, "reward_total_composite_std": 0.2603684365749359} {"timestamp_utc": "2026-04-13T09:37:06Z", "mode": "train", "global_step": 899, "epoch": 0.09030637870416876, "loss": 0.0371, "grad_norm": 11.645864486694336, "learning_rate": 7.2787878787878795e-06, "num_tokens": 1588243.0, "completions/mean_length": 54.625, "completions/min_length": 49.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.625, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.8100303411483765, "rewards/meter/std": 0.32428187131881714, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9260385036468506, "rewards/repeat_soft/std": 0.07360906898975372, "rewards/judge_quality/mean": 0.7362500429153442, "rewards/judge_quality/std": 0.25376805663108826, "rewards/total_composite/mean": 0.7051819562911987, "rewards/total_composite/std": 0.19023671746253967, "reward": 0.7051819562911987, "reward_std": 0.19023671746253967, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16436229646205902, "sampling/sampling_logp_difference/max": 1.3809700012207031, "sampling/importance_sampling_ratio/min": 0.25133463740348816, "sampling/importance_sampling_ratio/mean": 1.0121458768844604, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1463303491473198, "clip_ratio/low_mean": 0.12049462832510471, "clip_ratio/low_min": 0.12049462832510471, "clip_ratio/high_mean": 0.052040016278624535, "clip_ratio/high_max": 0.052040016278624535, "clip_ratio/region_mean": 0.17253464460372925, "reward_total_mean": 0.7051819562911987, "reward_meter_mean": 0.8100303411483765, "reward_meter_std": 0.32428187131881714, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9260385036468506, "reward_repeat_soft_std": 0.07360906898975372, "reward_judge_quality_mean": 0.7362500429153442, "reward_judge_quality_std": 0.25376805663108826, "reward_total_composite_mean": 0.7051819562911987, "reward_total_composite_std": 0.19023671746253967} {"timestamp_utc": "2026-04-13T09:37:12Z", "mode": "train", "global_step": 900, "epoch": 0.09040683073832245, "loss": 0.0353, "grad_norm": 7.044389724731445, "learning_rate": 7.275757575757576e-06, "num_tokens": 1590605.0, "completions/mean_length": 107.25, "completions/min_length": 92.0, "completions/max_length": 122.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.25, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 122.0, "rewards/meter/mean": 0.8630037307739258, "rewards/meter/std": 0.20947736501693726, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9194906949996948, "rewards/repeat_soft/std": 0.062074482440948486, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.5716500878334045, "rewards/total_composite/std": 0.12632188200950623, "reward": 0.5716500878334045, "reward_std": 0.12632188200950623, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13405700027942657, "sampling/sampling_logp_difference/max": 1.6423745155334473, "sampling/importance_sampling_ratio/min": 0.1935199648141861, "sampling/importance_sampling_ratio/mean": 1.0059252977371216, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0670483708381653, "clip_ratio/low_mean": 0.08348380029201508, "clip_ratio/low_min": 0.08348380029201508, "clip_ratio/high_mean": 0.02966239769011736, "clip_ratio/high_max": 0.02966239769011736, "clip_ratio/region_mean": 0.11314619798213243, "reward_total_mean": 0.5716500878334045, "reward_meter_mean": 0.8630037307739258, "reward_meter_std": 0.20947736501693726, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9194906949996948, "reward_repeat_soft_std": 0.062074482440948486, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.5716500878334045, "reward_total_composite_std": 0.12632188200950623} {"timestamp_utc": "2026-04-13T09:38:06Z", "mode": "eval", "global_step": 900, "epoch": 0.09040683073832245, "eval_loss": NaN, "eval_runtime": 53.4727, "eval_samples_per_second": 1.496, "eval_steps_per_second": 0.187, "eval_num_tokens": 1590605.0, "eval_completions/mean_length": 97.8125, "eval_completions/min_length": 36.7, "eval_completions/max_length": 239.6, "eval_completions/clipped_ratio": 0.05, "eval_completions/mean_terminated_length": 76.43095321655274, "eval_completions/min_terminated_length": 36.7, "eval_completions/max_terminated_length": 122.0, "eval_rewards/meter/mean": 0.7157208025455475, "eval_rewards/meter/std": 0.3429990842938423, "eval_rewards/count_adherence/mean": 0.9489583134651184, "eval_rewards/count_adherence/std": 0.0947537835687399, "eval_rewards/hard_gate/mean": 0.925, "eval_rewards/hard_gate/std": 0.18771235942840575, "eval_rewards/repeat_soft/mean": 0.9064767360687256, "eval_rewards/repeat_soft/std": 0.10691137947142124, "eval_rewards/judge_quality/mean": 0.48524999618530273, "eval_rewards/judge_quality/std": 0.19364597499370576, "eval_rewards/total_composite/mean": 0.5320614576339722, "eval_rewards/total_composite/std": 0.19630679711699486, "eval_reward": 0.5320614576339722, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.08199086189270019, "eval_sampling/sampling_logp_difference/max": 1.0795791625976563, "eval_sampling/importance_sampling_ratio/min": 0.3488084301352501, "eval_sampling/importance_sampling_ratio/mean": 1.0217031955718994, "eval_sampling/importance_sampling_ratio/max": 1.4728089570999146, "eval_entropy": 0.963804829120636, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5320614576339722, "eval_reward_meter_mean": 0.7157208025455475, "eval_reward_meter_std": 0.3429990842938423, "eval_reward_count_adherence_mean": 0.9489583134651184, "eval_reward_count_adherence_std": 0.0947537835687399, "eval_reward_hard_gate_mean": 0.925, "eval_reward_hard_gate_std": 0.18771235942840575, "eval_reward_repeat_soft_mean": 0.9064767360687256, "eval_reward_repeat_soft_std": 0.10691137947142124, "eval_reward_judge_quality_mean": 0.48524999618530273, "eval_reward_judge_quality_std": 0.19364597499370576, "eval_reward_total_composite_mean": 0.5320614576339722, "eval_reward_total_composite_std": 0.19630679711699486} {"timestamp_utc": "2026-04-13T09:38:15Z", "mode": "train", "global_step": 901, "epoch": 0.09050728277247615, "loss": 0.0298, "grad_norm": 12.812335968017578, "learning_rate": 7.272727272727273e-06, "num_tokens": 1592260.0, "completions/mean_length": 46.875, "completions/min_length": 40.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.875, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.08536005020141602, "rewards/meter/std": 0.08806665241718292, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7610586881637573, "rewards/repeat_soft/std": 0.18405461311340332, "rewards/judge_quality/mean": 0.6449999809265137, "rewards/judge_quality/std": 0.24928471446037292, "rewards/total_composite/mean": 0.3542100787162781, "rewards/total_composite/std": 0.06443716585636139, "reward": 0.3542100787162781, "reward_std": 0.06443716585636139, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13556799292564392, "sampling/sampling_logp_difference/max": 1.5642199516296387, "sampling/importance_sampling_ratio/min": 0.20925118029117584, "sampling/importance_sampling_ratio/mean": 1.0086678266525269, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7705042622983456, "clip_ratio/low_mean": 0.051945476327091455, "clip_ratio/low_min": 0.051945476327091455, "clip_ratio/high_mean": 0.100767582654953, "clip_ratio/high_max": 0.100767582654953, "clip_ratio/region_mean": 0.15271305898204446, "reward_total_mean": 0.3542100787162781, "reward_meter_mean": 0.08536005020141602, "reward_meter_std": 0.08806665241718292, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7610586881637573, "reward_repeat_soft_std": 0.18405461311340332, "reward_judge_quality_mean": 0.6449999809265137, "reward_judge_quality_std": 0.24928471446037292, "reward_total_composite_mean": 0.3542100787162781, "reward_total_composite_std": 0.06443716585636139} {"timestamp_utc": "2026-04-13T09:38:22Z", "mode": "train", "global_step": 902, "epoch": 0.09060773480662983, "loss": 0.0679, "grad_norm": 14.534664154052734, "learning_rate": 7.26969696969697e-06, "num_tokens": 1593970.0, "completions/mean_length": 56.75, "completions/min_length": 39.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.75, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.4406425356864929, "rewards/meter/std": 0.4224933087825775, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9830940365791321, "rewards/repeat_soft/std": 0.013649843633174896, "rewards/judge_quality/mean": 0.5062500238418579, "rewards/judge_quality/std": 0.13265828788280487, "rewards/total_composite/mean": 0.49974721670150757, "rewards/total_composite/std": 0.16015437245368958, "reward": 0.49974721670150757, "reward_std": 0.16015438735485077, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1860671043395996, "sampling/sampling_logp_difference/max": 3.8892030715942383, "sampling/importance_sampling_ratio/min": 0.02046164683997631, "sampling/importance_sampling_ratio/mean": 1.0121606588363647, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9524694234132767, "clip_ratio/low_mean": 0.11109669506549835, "clip_ratio/low_min": 0.11109669506549835, "clip_ratio/high_mean": 0.05636154394596815, "clip_ratio/high_max": 0.05636154394596815, "clip_ratio/region_mean": 0.1674582390114665, "reward_total_mean": 0.49974721670150757, "reward_meter_mean": 0.4406425356864929, "reward_meter_std": 0.4224933087825775, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9830940365791321, "reward_repeat_soft_std": 0.013649843633174896, "reward_judge_quality_mean": 0.5062500238418579, "reward_judge_quality_std": 0.13265828788280487, "reward_total_composite_mean": 0.49974721670150757, "reward_total_composite_std": 0.16015437245368958} {"timestamp_utc": "2026-04-13T09:38:34Z", "mode": "train", "global_step": 903, "epoch": 0.09070818684078352, "loss": -0.1015, "grad_norm": 3.356720447540283, "learning_rate": 7.266666666666668e-06, "num_tokens": 1595627.0, "completions/mean_length": 95.125, "completions/min_length": 30.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 35.57143020629883, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.6811458468437195, "rewards/meter/std": 0.3611838221549988, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9799220561981201, "rewards/repeat_soft/std": 0.016584882512688637, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.23439893126487732, "rewards/total_composite/mean": 0.4835904538631439, "rewards/total_composite/std": 0.21161921322345734, "reward": 0.4835904538631439, "reward_std": 0.21161922812461853, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16999176144599915, "sampling/sampling_logp_difference/max": 2.0852737426757812, "sampling/importance_sampling_ratio/min": 0.12427309155464172, "sampling/importance_sampling_ratio/mean": 1.0145553350448608, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8613782152533531, "clip_ratio/low_mean": 0.015243902802467346, "clip_ratio/low_min": 0.015243902802467346, "clip_ratio/high_mean": 0.10294332588091493, "clip_ratio/high_max": 0.10294332588091493, "clip_ratio/region_mean": 0.11818722868338227, "reward_total_mean": 0.4835904538631439, "reward_meter_mean": 0.6811458468437195, "reward_meter_std": 0.3611838221549988, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9799220561981201, "reward_repeat_soft_std": 0.016584882512688637, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.23439893126487732, "reward_total_composite_mean": 0.4835904538631439, "reward_total_composite_std": 0.21161921322345734} {"timestamp_utc": "2026-04-13T09:38:40Z", "mode": "train", "global_step": 904, "epoch": 0.09080863887493722, "loss": 0.0736, "grad_norm": 14.554370880126953, "learning_rate": 7.263636363636364e-06, "num_tokens": 1597503.0, "completions/mean_length": 64.5, "completions/min_length": 57.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.5, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.778179407119751, "rewards/meter/std": 0.2762916684150696, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.877153754234314, "rewards/repeat_soft/std": 0.18072569370269775, "rewards/judge_quality/mean": 0.5649999976158142, "rewards/judge_quality/std": 0.194054514169693, "rewards/total_composite/mean": 0.5995876789093018, "rewards/total_composite/std": 0.10906385630369186, "reward": 0.5995876789093018, "reward_std": 0.10906385630369186, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1650829166173935, "sampling/sampling_logp_difference/max": 1.500819206237793, "sampling/importance_sampling_ratio/min": 0.24453456699848175, "sampling/importance_sampling_ratio/mean": 1.0208650827407837, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.057773917913437, "clip_ratio/low_mean": 0.08504586480557919, "clip_ratio/low_min": 0.08504586480557919, "clip_ratio/high_mean": 0.0739332064986229, "clip_ratio/high_max": 0.0739332064986229, "clip_ratio/region_mean": 0.15897907130420208, "reward_total_mean": 0.5995876789093018, "reward_meter_mean": 0.778179407119751, "reward_meter_std": 0.2762916684150696, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.877153754234314, "reward_repeat_soft_std": 0.18072569370269775, "reward_judge_quality_mean": 0.5649999976158142, "reward_judge_quality_std": 0.194054514169693, "reward_total_composite_mean": 0.5995876789093018, "reward_total_composite_std": 0.10906385630369186} {"timestamp_utc": "2026-04-13T09:38:46Z", "mode": "train", "global_step": 905, "epoch": 0.09090909090909091, "loss": 0.0897, "grad_norm": 18.605655670166016, "learning_rate": 7.260606060606061e-06, "num_tokens": 1598909.0, "completions/mean_length": 24.75, "completions/min_length": 21.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.75, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.8512672185897827, "rewards/meter/std": 0.25897178053855896, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9602519273757935, "rewards/repeat_soft/std": 0.006358357612043619, "rewards/judge_quality/mean": 0.7437499761581421, "rewards/judge_quality/std": 0.2432481348514557, "rewards/total_composite/mean": 0.7538845539093018, "rewards/total_composite/std": 0.19257384538650513, "reward": 0.7538845539093018, "reward_std": 0.19257384538650513, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12659122049808502, "sampling/sampling_logp_difference/max": 1.3429498672485352, "sampling/importance_sampling_ratio/min": 0.26107439398765564, "sampling/importance_sampling_ratio/mean": 1.024914026260376, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6850458532571793, "clip_ratio/low_mean": 0.06517857313156128, "clip_ratio/low_min": 0.06517857313156128, "clip_ratio/high_mean": 0.07068452564999461, "clip_ratio/high_max": 0.07068452564999461, "clip_ratio/region_mean": 0.1358630987815559, "reward_total_mean": 0.7538845539093018, "reward_meter_mean": 0.8512672185897827, "reward_meter_std": 0.25897178053855896, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9602519273757935, "reward_repeat_soft_std": 0.006358357612043619, "reward_judge_quality_mean": 0.7437499761581421, "reward_judge_quality_std": 0.2432481348514557, "reward_total_composite_mean": 0.7538845539093018, "reward_total_composite_std": 0.19257384538650513} {"timestamp_utc": "2026-04-13T09:38:57Z", "mode": "train", "global_step": 906, "epoch": 0.0910095429432446, "loss": -0.1552, "grad_norm": 3.5159833431243896, "learning_rate": 7.257575757575758e-06, "num_tokens": 1601376.0, "completions/mean_length": 166.375, "completions/min_length": 102.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 117.00000762939453, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.703851044178009, "rewards/meter/std": 0.32434767484664917, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9248175024986267, "rewards/repeat_soft/std": 0.07298389822244644, "rewards/judge_quality/mean": 0.6112500429153442, "rewards/judge_quality/std": 0.3435710668563843, "rewards/total_composite/mean": 0.598581850528717, "rewards/total_composite/std": 0.291879266500473, "reward": 0.598581850528717, "reward_std": 0.291879266500473, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13965491950511932, "sampling/sampling_logp_difference/max": 1.4687843322753906, "sampling/importance_sampling_ratio/min": 0.23020517826080322, "sampling/importance_sampling_ratio/mean": 1.0273159742355347, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9906918928027153, "clip_ratio/low_mean": 0.02160807466134429, "clip_ratio/low_min": 0.02160807466134429, "clip_ratio/high_mean": 0.08824530150741339, "clip_ratio/high_max": 0.08824530150741339, "clip_ratio/region_mean": 0.10985337616875768, "reward_total_mean": 0.598581850528717, "reward_meter_mean": 0.703851044178009, "reward_meter_std": 0.32434767484664917, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9248175024986267, "reward_repeat_soft_std": 0.07298389822244644, "reward_judge_quality_mean": 0.6112500429153442, "reward_judge_quality_std": 0.3435710668563843, "reward_total_composite_mean": 0.598581850528717, "reward_total_composite_std": 0.291879266500473} {"timestamp_utc": "2026-04-13T09:39:04Z", "mode": "train", "global_step": 907, "epoch": 0.09110999497739829, "loss": 0.0255, "grad_norm": 7.501486778259277, "learning_rate": 7.254545454545455e-06, "num_tokens": 1603558.0, "completions/mean_length": 104.75, "completions/min_length": 96.0, "completions/max_length": 118.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 104.75, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 118.0, "rewards/meter/mean": 0.8907060623168945, "rewards/meter/std": 0.26040831208229065, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7417531609535217, "rewards/repeat_soft/std": 0.1804305613040924, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.12906257808208466, "rewards/total_composite/mean": 0.5654819011688232, "rewards/total_composite/std": 0.1330239325761795, "reward": 0.5654819011688232, "reward_std": 0.1330239474773407, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10251452028751373, "sampling/sampling_logp_difference/max": 1.8215961456298828, "sampling/importance_sampling_ratio/min": 0.16176734864711761, "sampling/importance_sampling_ratio/mean": 1.0123924016952515, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7758409716188908, "clip_ratio/low_mean": 0.022410391829907894, "clip_ratio/low_min": 0.022410391829907894, "clip_ratio/high_mean": 0.08412217628210783, "clip_ratio/high_max": 0.08412217628210783, "clip_ratio/region_mean": 0.10653256811201572, "reward_total_mean": 0.5654819011688232, "reward_meter_mean": 0.8907060623168945, "reward_meter_std": 0.26040831208229065, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7417531609535217, "reward_repeat_soft_std": 0.1804305613040924, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.12906257808208466, "reward_total_composite_mean": 0.5654819011688232, "reward_total_composite_std": 0.1330239325761795} {"timestamp_utc": "2026-04-13T09:39:11Z", "mode": "train", "global_step": 908, "epoch": 0.09121044701155198, "loss": 0.0777, "grad_norm": 9.802550315856934, "learning_rate": 7.251515151515151e-06, "num_tokens": 1605881.0, "completions/mean_length": 106.375, "completions/min_length": 87.0, "completions/max_length": 126.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.375, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 126.0, "rewards/meter/mean": 0.9819270372390747, "rewards/meter/std": 0.01999370940029621, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9211548566818237, "rewards/repeat_soft/std": 0.050338804721832275, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.7271592617034912, "rewards/total_composite/std": 0.16884341835975647, "reward": 0.7271592617034912, "reward_std": 0.16884341835975647, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16331268846988678, "sampling/sampling_logp_difference/max": 1.952540397644043, "sampling/importance_sampling_ratio/min": 0.14191310107707977, "sampling/importance_sampling_ratio/mean": 1.0145978927612305, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2415838986635208, "clip_ratio/low_mean": 0.08216759376227856, "clip_ratio/low_min": 0.08216759376227856, "clip_ratio/high_mean": 0.05946306139230728, "clip_ratio/high_max": 0.05946306139230728, "clip_ratio/region_mean": 0.14163065515458584, "reward_total_mean": 0.7271592617034912, "reward_meter_mean": 0.9819270372390747, "reward_meter_std": 0.01999370940029621, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9211548566818237, "reward_repeat_soft_std": 0.050338804721832275, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.7271592617034912, "reward_total_composite_std": 0.16884341835975647} {"timestamp_utc": "2026-04-13T09:39:17Z", "mode": "train", "global_step": 909, "epoch": 0.09131089904570568, "loss": 0.0426, "grad_norm": 18.87795066833496, "learning_rate": 7.2484848484848495e-06, "num_tokens": 1607426.0, "completions/mean_length": 34.125, "completions/min_length": 30.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.125, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.33157655596733093, "rewards/meter/std": 0.31648188829421997, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9801974296569824, "rewards/repeat_soft/std": 0.012491202913224697, "rewards/judge_quality/mean": 0.9200000166893005, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5453124046325684, "rewards/total_composite/std": 0.19037026166915894, "reward": 0.5453124046325684, "reward_std": 0.19037026166915894, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16925941407680511, "sampling/sampling_logp_difference/max": 1.7289962768554688, "sampling/importance_sampling_ratio/min": 0.17746244370937347, "sampling/importance_sampling_ratio/mean": 1.0000395774841309, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7677922323346138, "clip_ratio/low_mean": 0.06933283852413297, "clip_ratio/low_min": 0.06933283852413297, "clip_ratio/high_mean": 0.05304053891450167, "clip_ratio/high_max": 0.05304053891450167, "clip_ratio/region_mean": 0.12237337743863463, "reward_total_mean": 0.5453124046325684, "reward_meter_mean": 0.33157655596733093, "reward_meter_std": 0.31648188829421997, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9801974296569824, "reward_repeat_soft_std": 0.012491202913224697, "reward_judge_quality_mean": 0.9200000166893005, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5453124046325684, "reward_total_composite_std": 0.19037026166915894} {"timestamp_utc": "2026-04-13T09:39:28Z", "mode": "train", "global_step": 910, "epoch": 0.09141135107985937, "loss": -0.1004, "grad_norm": 3.7054879665374756, "learning_rate": 7.245454545454546e-06, "num_tokens": 1609176.0, "completions/mean_length": 110.75, "completions/min_length": 47.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 53.42857360839844, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.6746935844421387, "rewards/meter/std": 0.38369810581207275, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9529876112937927, "rewards/repeat_soft/std": 0.03730596601963043, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.23445606231689453, "rewards/total_composite/mean": 0.5209311246871948, "rewards/total_composite/std": 0.25986728072166443, "reward": 0.5209311246871948, "reward_std": 0.25986728072166443, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1810074895620346, "sampling/sampling_logp_difference/max": 1.452235221862793, "sampling/importance_sampling_ratio/min": 0.2340465486049652, "sampling/importance_sampling_ratio/mean": 1.0295379161834717, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4939635396003723, "clip_ratio/low_mean": 0.03707153536379337, "clip_ratio/low_min": 0.03707153536379337, "clip_ratio/high_mean": 0.0893766162917018, "clip_ratio/high_max": 0.0893766162917018, "clip_ratio/region_mean": 0.12644815165549517, "reward_total_mean": 0.5209311246871948, "reward_meter_mean": 0.6746935844421387, "reward_meter_std": 0.38369810581207275, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9529876112937927, "reward_repeat_soft_std": 0.03730596601963043, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.23445606231689453, "reward_total_composite_mean": 0.5209311246871948, "reward_total_composite_std": 0.25986728072166443} {"timestamp_utc": "2026-04-13T09:39:34Z", "mode": "train", "global_step": 911, "epoch": 0.09151180311401307, "loss": 0.0266, "grad_norm": 14.23844051361084, "learning_rate": 7.242424242424243e-06, "num_tokens": 1610922.0, "completions/mean_length": 39.25, "completions/min_length": 34.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.25, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.1473059356212616, "rewards/meter/std": 0.3147366940975189, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.7999280691146851, "rewards/repeat_soft/std": 0.11304853856563568, "rewards/judge_quality/mean": 0.7325000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.3703829348087311, "rewards/total_composite/std": 0.2418767809867859, "reward": 0.3703829348087311, "reward_std": 0.2418767660856247, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1460590809583664, "sampling/sampling_logp_difference/max": 2.1889023780822754, "sampling/importance_sampling_ratio/min": 0.11203965544700623, "sampling/importance_sampling_ratio/mean": 0.9989128708839417, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4969974309206009, "clip_ratio/low_mean": 0.10410348512232304, "clip_ratio/low_min": 0.10410348512232304, "clip_ratio/high_mean": 0.03947368450462818, "clip_ratio/high_max": 0.03947368450462818, "clip_ratio/region_mean": 0.14357716962695122, "reward_total_mean": 0.3703829348087311, "reward_meter_mean": 0.1473059356212616, "reward_meter_std": 0.3147366940975189, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.7999280691146851, "reward_repeat_soft_std": 0.11304853856563568, "reward_judge_quality_mean": 0.7325000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.3703829348087311, "reward_total_composite_std": 0.2418767809867859} {"timestamp_utc": "2026-04-13T09:39:40Z", "mode": "train", "global_step": 912, "epoch": 0.09161225514816675, "loss": 0.09, "grad_norm": 20.343666076660156, "learning_rate": 7.2393939393939404e-06, "num_tokens": 1612603.0, "completions/mean_length": 36.125, "completions/min_length": 30.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.125, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.5597535371780396, "rewards/meter/std": 0.3825490474700928, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9814321398735046, "rewards/repeat_soft/std": 0.018087225034832954, "rewards/judge_quality/mean": 0.4362499713897705, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.5003030300140381, "rewards/total_composite/std": 0.10381560027599335, "reward": 0.5003030300140381, "reward_std": 0.10381558537483215, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17298763990402222, "sampling/sampling_logp_difference/max": 1.3021645545959473, "sampling/importance_sampling_ratio/min": 0.27194252610206604, "sampling/importance_sampling_ratio/mean": 1.0064977407455444, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.075918272137642, "clip_ratio/low_mean": 0.08903274312615395, "clip_ratio/low_min": 0.08903274312615395, "clip_ratio/high_mean": 0.07369153341278434, "clip_ratio/high_max": 0.07369153341278434, "clip_ratio/region_mean": 0.16272427653893828, "reward_total_mean": 0.5003030300140381, "reward_meter_mean": 0.5597535371780396, "reward_meter_std": 0.3825490474700928, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9814321398735046, "reward_repeat_soft_std": 0.018087225034832954, "reward_judge_quality_mean": 0.4362499713897705, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.5003030300140381, "reward_total_composite_std": 0.10381560027599335} {"timestamp_utc": "2026-04-13T09:39:46Z", "mode": "train", "global_step": 913, "epoch": 0.09171270718232044, "loss": 0.0452, "grad_norm": 14.120256423950195, "learning_rate": 7.236363636363637e-06, "num_tokens": 1614162.0, "completions/mean_length": 46.875, "completions/min_length": 38.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.875, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.8159139752388, "rewards/meter/std": 0.24263866245746613, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9638544321060181, "rewards/repeat_soft/std": 0.030508361756801605, "rewards/judge_quality/mean": 0.643750011920929, "rewards/judge_quality/std": 0.18133927881717682, "rewards/total_composite/mean": 0.6741479635238647, "rewards/total_composite/std": 0.12217270582914352, "reward": 0.6741479635238647, "reward_std": 0.12217270582914352, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1827840507030487, "sampling/sampling_logp_difference/max": 1.6252827644348145, "sampling/importance_sampling_ratio/min": 0.21006464958190918, "sampling/importance_sampling_ratio/mean": 1.019508957862854, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4227289482951164, "clip_ratio/low_mean": 0.10298752877861261, "clip_ratio/low_min": 0.10298752877861261, "clip_ratio/high_mean": 0.08171163313090801, "clip_ratio/high_max": 0.08171163313090801, "clip_ratio/region_mean": 0.18469916190952063, "reward_total_mean": 0.6741479635238647, "reward_meter_mean": 0.8159139752388, "reward_meter_std": 0.24263866245746613, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9638544321060181, "reward_repeat_soft_std": 0.030508361756801605, "reward_judge_quality_mean": 0.643750011920929, "reward_judge_quality_std": 0.18133927881717682, "reward_total_composite_mean": 0.6741479635238647, "reward_total_composite_std": 0.12217270582914352} {"timestamp_utc": "2026-04-13T09:39:53Z", "mode": "train", "global_step": 914, "epoch": 0.09181315921647414, "loss": 0.0263, "grad_norm": 12.239043235778809, "learning_rate": 7.233333333333334e-06, "num_tokens": 1615772.0, "completions/mean_length": 47.25, "completions/min_length": 46.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.25, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.9544693827629089, "rewards/meter/std": 0.05370640382170677, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.966535747051239, "rewards/repeat_soft/std": 0.016960065811872482, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.17136012017726898, "rewards/total_composite/mean": 0.6542842388153076, "rewards/total_composite/std": 0.11274627596139908, "reward": 0.6542842388153076, "reward_std": 0.11274627596139908, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11110243201255798, "sampling/sampling_logp_difference/max": 1.6137062311172485, "sampling/importance_sampling_ratio/min": 0.19914816319942474, "sampling/importance_sampling_ratio/mean": 1.0147902965545654, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.70686886459589, "clip_ratio/low_mean": 0.06612967816181481, "clip_ratio/low_min": 0.06612967816181481, "clip_ratio/high_mean": 0.02445652149617672, "clip_ratio/high_max": 0.02445652149617672, "clip_ratio/region_mean": 0.09058619965799153, "reward_total_mean": 0.6542842388153076, "reward_meter_mean": 0.9544693827629089, "reward_meter_std": 0.05370640382170677, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.966535747051239, "reward_repeat_soft_std": 0.016960065811872482, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.17136012017726898, "reward_total_composite_mean": 0.6542842388153076, "reward_total_composite_std": 0.11274627596139908} {"timestamp_utc": "2026-04-13T09:40:00Z", "mode": "train", "global_step": 915, "epoch": 0.09191361125062783, "loss": 0.0788, "grad_norm": 9.88178539276123, "learning_rate": 7.2303030303030305e-06, "num_tokens": 1617758.0, "completions/mean_length": 85.25, "completions/min_length": 76.0, "completions/max_length": 92.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.25, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 92.0, "rewards/meter/mean": 0.672178328037262, "rewards/meter/std": 0.30288252234458923, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8553141355514526, "rewards/repeat_soft/std": 0.09193862974643707, "rewards/judge_quality/mean": 0.5824999809265137, "rewards/judge_quality/std": 0.23260943591594696, "rewards/total_composite/mean": 0.5740668773651123, "rewards/total_composite/std": 0.15459750592708588, "reward": 0.5740668773651123, "reward_std": 0.15459749102592468, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16788269579410553, "sampling/sampling_logp_difference/max": 3.2447891235351562, "sampling/importance_sampling_ratio/min": 0.03897678107023239, "sampling/importance_sampling_ratio/mean": 0.9828201532363892, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6544415801763535, "clip_ratio/low_mean": 0.07083723042160273, "clip_ratio/low_min": 0.07083723042160273, "clip_ratio/high_mean": 0.07165043521672487, "clip_ratio/high_max": 0.07165043521672487, "clip_ratio/region_mean": 0.1424876656383276, "reward_total_mean": 0.5740668773651123, "reward_meter_mean": 0.672178328037262, "reward_meter_std": 0.30288252234458923, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8553141355514526, "reward_repeat_soft_std": 0.09193862974643707, "reward_judge_quality_mean": 0.5824999809265137, "reward_judge_quality_std": 0.23260943591594696, "reward_total_composite_mean": 0.5740668773651123, "reward_total_composite_std": 0.15459750592708588} {"timestamp_utc": "2026-04-13T09:40:06Z", "mode": "train", "global_step": 916, "epoch": 0.09201406328478151, "loss": 0.0469, "grad_norm": 20.423137664794922, "learning_rate": 7.227272727272729e-06, "num_tokens": 1619310.0, "completions/mean_length": 33.0, "completions/min_length": 31.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.0, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.5603809356689453, "rewards/meter/std": 0.36889195442199707, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9368240237236023, "rewards/repeat_soft/std": 0.0935758501291275, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.22720350325107574, "rewards/total_composite/mean": 0.5057446956634521, "rewards/total_composite/std": 0.08553271740674973, "reward": 0.5057446956634521, "reward_std": 0.08553270995616913, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17337189614772797, "sampling/sampling_logp_difference/max": 4.319033145904541, "sampling/importance_sampling_ratio/min": 0.0133127486333251, "sampling/importance_sampling_ratio/mean": 0.9902936816215515, "sampling/importance_sampling_ratio/max": 1.9824931621551514, "entropy": 0.8912190422415733, "clip_ratio/low_mean": 0.06076621077954769, "clip_ratio/low_min": 0.06076621077954769, "clip_ratio/high_mean": 0.09892127197235823, "clip_ratio/high_max": 0.09892127197235823, "clip_ratio/region_mean": 0.15968748275190592, "reward_total_mean": 0.5057446956634521, "reward_meter_mean": 0.5603809356689453, "reward_meter_std": 0.36889195442199707, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9368240237236023, "reward_repeat_soft_std": 0.0935758501291275, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.22720350325107574, "reward_total_composite_mean": 0.5057446956634521, "reward_total_composite_std": 0.08553271740674973} {"timestamp_utc": "2026-04-13T09:40:17Z", "mode": "train", "global_step": 917, "epoch": 0.0921145153189352, "loss": -0.0847, "grad_norm": 4.606402397155762, "learning_rate": 7.224242424242425e-06, "num_tokens": 1620756.0, "completions/mean_length": 104.75, "completions/min_length": 42.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 46.57143020629883, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.9420249462127686, "rewards/meter/std": 0.08945825695991516, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9198586940765381, "rewards/repeat_soft/std": 0.018826227635145187, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.13617216050624847, "rewards/total_composite/mean": 0.45015519857406616, "rewards/total_composite/std": 0.279334157705307, "reward": 0.45015519857406616, "reward_std": 0.279334157705307, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12947139143943787, "sampling/sampling_logp_difference/max": 1.295595407485962, "sampling/importance_sampling_ratio/min": 0.27373483777046204, "sampling/importance_sampling_ratio/mean": 1.0317928791046143, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9875844419002533, "clip_ratio/low_mean": 0.007978723384439945, "clip_ratio/low_min": 0.007978723384439945, "clip_ratio/high_mean": 0.08037575893104076, "clip_ratio/high_max": 0.08037575893104076, "clip_ratio/region_mean": 0.08835448231548071, "reward_total_mean": 0.45015519857406616, "reward_meter_mean": 0.9420249462127686, "reward_meter_std": 0.08945825695991516, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9198586940765381, "reward_repeat_soft_std": 0.018826227635145187, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.13617216050624847, "reward_total_composite_mean": 0.45015519857406616, "reward_total_composite_std": 0.279334157705307} {"timestamp_utc": "2026-04-13T09:40:28Z", "mode": "train", "global_step": 918, "epoch": 0.0922149673530889, "loss": -0.143, "grad_norm": 5.479053974151611, "learning_rate": 7.221212121212122e-06, "num_tokens": 1622431.0, "completions/mean_length": 121.375, "completions/min_length": 60.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 65.5714340209961, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.7508893609046936, "rewards/meter/std": 0.2547018229961395, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9491066932678223, "rewards/repeat_soft/std": 0.02991720475256443, "rewards/judge_quality/mean": 0.3787499964237213, "rewards/judge_quality/std": 0.2713688910007477, "rewards/total_composite/mean": 0.4926729202270508, "rewards/total_composite/std": 0.25319477915763855, "reward": 0.4926729202270508, "reward_std": 0.25319477915763855, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16436295211315155, "sampling/sampling_logp_difference/max": 1.2689733505249023, "sampling/importance_sampling_ratio/min": 0.28112009167671204, "sampling/importance_sampling_ratio/mean": 1.0305107831954956, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.070231020450592, "clip_ratio/low_mean": 0.06362179573625326, "clip_ratio/low_min": 0.06362179573625326, "clip_ratio/high_mean": 0.07916897907853127, "clip_ratio/high_max": 0.07916897907853127, "clip_ratio/region_mean": 0.14279077481478453, "reward_total_mean": 0.4926729202270508, "reward_meter_mean": 0.7508893609046936, "reward_meter_std": 0.2547018229961395, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9491066932678223, "reward_repeat_soft_std": 0.02991720475256443, "reward_judge_quality_mean": 0.3787499964237213, "reward_judge_quality_std": 0.2713688910007477, "reward_total_composite_mean": 0.4926729202270508, "reward_total_composite_std": 0.25319477915763855} {"timestamp_utc": "2026-04-13T09:40:34Z", "mode": "train", "global_step": 919, "epoch": 0.0923154193872426, "loss": 0.0613, "grad_norm": 14.65757942199707, "learning_rate": 7.218181818181819e-06, "num_tokens": 1623905.0, "completions/mean_length": 43.25, "completions/min_length": 40.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.25, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.9844621419906616, "rewards/meter/std": 0.010709667578339577, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9381470680236816, "rewards/repeat_soft/std": 0.046228960156440735, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.6191074252128601, "rewards/total_composite/std": 0.015671368688344955, "reward": 0.6191074252128601, "reward_std": 0.015671366825699806, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16086256504058838, "sampling/sampling_logp_difference/max": 1.6122045516967773, "sampling/importance_sampling_ratio/min": 0.1994474232196808, "sampling/importance_sampling_ratio/mean": 0.9934461116790771, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8492658138275146, "clip_ratio/low_mean": 0.07349553424865007, "clip_ratio/low_min": 0.07349553424865007, "clip_ratio/high_mean": 0.0476880082860589, "clip_ratio/high_max": 0.0476880082860589, "clip_ratio/region_mean": 0.12118354253470898, "reward_total_mean": 0.6191074252128601, "reward_meter_mean": 0.9844621419906616, "reward_meter_std": 0.010709667578339577, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9381470680236816, "reward_repeat_soft_std": 0.046228960156440735, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.6191074252128601, "reward_total_composite_std": 0.015671368688344955} {"timestamp_utc": "2026-04-13T09:40:45Z", "mode": "train", "global_step": 920, "epoch": 0.09241587142139629, "loss": -0.1817, "grad_norm": 2.979152202606201, "learning_rate": 7.215151515151516e-06, "num_tokens": 1625977.0, "completions/mean_length": 155.0, "completions/min_length": 96.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 104.00000762939453, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 113.0, "rewards/meter/mean": 0.767349362373352, "rewards/meter/std": 0.2601783275604248, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9513424634933472, "rewards/repeat_soft/std": 0.03866739943623543, "rewards/judge_quality/mean": 0.4112499952316284, "rewards/judge_quality/std": 0.1797965168952942, "rewards/total_composite/mean": 0.5252867341041565, "rewards/total_composite/std": 0.23032616078853607, "reward": 0.5252867341041565, "reward_std": 0.23032614588737488, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14671450853347778, "sampling/sampling_logp_difference/max": 1.4239892959594727, "sampling/importance_sampling_ratio/min": 0.24075166881084442, "sampling/importance_sampling_ratio/mean": 1.009709119796753, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0711248964071274, "clip_ratio/low_mean": 0.025735294446349144, "clip_ratio/low_min": 0.025735294446349144, "clip_ratio/high_mean": 0.11989929620176554, "clip_ratio/high_max": 0.11989929620176554, "clip_ratio/region_mean": 0.14563459064811468, "reward_total_mean": 0.5252867341041565, "reward_meter_mean": 0.767349362373352, "reward_meter_std": 0.2601783275604248, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9513424634933472, "reward_repeat_soft_std": 0.03866739943623543, "reward_judge_quality_mean": 0.4112499952316284, "reward_judge_quality_std": 0.1797965168952942, "reward_total_composite_mean": 0.5252867341041565, "reward_total_composite_std": 0.23032616078853607} {"timestamp_utc": "2026-04-13T09:40:52Z", "mode": "train", "global_step": 921, "epoch": 0.09251632345554997, "loss": -0.0443, "grad_norm": 12.08022403717041, "learning_rate": 7.212121212121212e-06, "num_tokens": 1627581.0, "completions/mean_length": 47.5, "completions/min_length": 40.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.5, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.6915476322174072, "rewards/meter/std": 0.36313360929489136, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9853203296661377, "rewards/repeat_soft/std": 0.022112060338258743, "rewards/judge_quality/mean": 0.6737500429153442, "rewards/judge_quality/std": 0.263435423374176, "rewards/total_composite/mean": 0.6625669002532959, "rewards/total_composite/std": 0.22150346636772156, "reward": 0.6625669002532959, "reward_std": 0.22150346636772156, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1407664716243744, "sampling/sampling_logp_difference/max": 2.5118842124938965, "sampling/importance_sampling_ratio/min": 0.0811152532696724, "sampling/importance_sampling_ratio/mean": 1.0195248126983643, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0722940489649773, "clip_ratio/low_mean": 0.10812500026077032, "clip_ratio/low_min": 0.10812500026077032, "clip_ratio/high_mean": 0.04378216154873371, "clip_ratio/high_max": 0.04378216154873371, "clip_ratio/region_mean": 0.15190716180950403, "reward_total_mean": 0.6625669002532959, "reward_meter_mean": 0.6915476322174072, "reward_meter_std": 0.36313360929489136, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9853203296661377, "reward_repeat_soft_std": 0.022112060338258743, "reward_judge_quality_mean": 0.6737500429153442, "reward_judge_quality_std": 0.263435423374176, "reward_total_composite_mean": 0.6625669002532959, "reward_total_composite_std": 0.22150346636772156} {"timestamp_utc": "2026-04-13T09:40:58Z", "mode": "train", "global_step": 922, "epoch": 0.09261677548970366, "loss": 0.0334, "grad_norm": 15.846829414367676, "learning_rate": 7.2090909090909104e-06, "num_tokens": 1629608.0, "completions/mean_length": 87.375, "completions/min_length": 76.0, "completions/max_length": 97.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.375, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.6068453788757324, "rewards/meter/std": 0.21214169263839722, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9070358276367188, "rewards/repeat_soft/std": 0.07087033987045288, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.5358442068099976, "rewards/total_composite/std": 0.13801956176757812, "reward": 0.5358442068099976, "reward_std": 0.13801954686641693, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17047323286533356, "sampling/sampling_logp_difference/max": 3.229947566986084, "sampling/importance_sampling_ratio/min": 0.03955957293510437, "sampling/importance_sampling_ratio/mean": 0.9821562767028809, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6564752086997032, "clip_ratio/low_mean": 0.08379318378865719, "clip_ratio/low_min": 0.08379318378865719, "clip_ratio/high_mean": 0.05765899270772934, "clip_ratio/high_max": 0.05765899270772934, "clip_ratio/region_mean": 0.14145217649638653, "reward_total_mean": 0.5358442068099976, "reward_meter_mean": 0.6068453788757324, "reward_meter_std": 0.21214169263839722, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9070358276367188, "reward_repeat_soft_std": 0.07087033987045288, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.5358442068099976, "reward_total_composite_std": 0.13801956176757812} {"timestamp_utc": "2026-04-13T09:41:04Z", "mode": "train", "global_step": 923, "epoch": 0.09271722752385736, "loss": 0.0264, "grad_norm": 14.256244659423828, "learning_rate": 7.206060606060606e-06, "num_tokens": 1631080.0, "completions/mean_length": 47.0, "completions/min_length": 39.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.0, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.5419894456863403, "rewards/meter/std": 0.36938440799713135, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9782922267913818, "rewards/repeat_soft/std": 0.02354535646736622, "rewards/judge_quality/mean": 0.7100000381469727, "rewards/judge_quality/std": 0.1302744448184967, "rewards/total_composite/mean": 0.580936074256897, "rewards/total_composite/std": 0.15119551122188568, "reward": 0.580936074256897, "reward_std": 0.15119551122188568, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13457390666007996, "sampling/sampling_logp_difference/max": 1.823359489440918, "sampling/importance_sampling_ratio/min": 0.16148234903812408, "sampling/importance_sampling_ratio/mean": 1.002490758895874, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.971940703690052, "clip_ratio/low_mean": 0.04533275682479143, "clip_ratio/low_min": 0.04533275682479143, "clip_ratio/high_mean": 0.11088750511407852, "clip_ratio/high_max": 0.11088750511407852, "clip_ratio/region_mean": 0.15622026193886995, "reward_total_mean": 0.580936074256897, "reward_meter_mean": 0.5419894456863403, "reward_meter_std": 0.36938440799713135, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9782922267913818, "reward_repeat_soft_std": 0.02354535646736622, "reward_judge_quality_mean": 0.7100000381469727, "reward_judge_quality_std": 0.1302744448184967, "reward_total_composite_mean": 0.580936074256897, "reward_total_composite_std": 0.15119551122188568} {"timestamp_utc": "2026-04-13T09:41:10Z", "mode": "train", "global_step": 924, "epoch": 0.09281767955801105, "loss": 0.0329, "grad_norm": 13.811881065368652, "learning_rate": 7.203030303030304e-06, "num_tokens": 1632629.0, "completions/mean_length": 42.625, "completions/min_length": 33.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.625, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.669313907623291, "rewards/meter/std": 0.3524772822856903, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9745993614196777, "rewards/repeat_soft/std": 0.02978009358048439, "rewards/judge_quality/mean": 0.4937500059604645, "rewards/judge_quality/std": 0.1728696972131729, "rewards/total_composite/mean": 0.5712888240814209, "rewards/total_composite/std": 0.1664804220199585, "reward": 0.5712888240814209, "reward_std": 0.1664804220199585, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1601911038160324, "sampling/sampling_logp_difference/max": 1.909719467163086, "sampling/importance_sampling_ratio/min": 0.1481219381093979, "sampling/importance_sampling_ratio/mean": 1.032793402671814, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0413456037640572, "clip_ratio/low_mean": 0.08925975020974874, "clip_ratio/low_min": 0.08925975020974874, "clip_ratio/high_mean": 0.07106302492320538, "clip_ratio/high_max": 0.07106302492320538, "clip_ratio/region_mean": 0.16032277513295412, "reward_total_mean": 0.5712888240814209, "reward_meter_mean": 0.669313907623291, "reward_meter_std": 0.3524772822856903, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9745993614196777, "reward_repeat_soft_std": 0.02978009358048439, "reward_judge_quality_mean": 0.4937500059604645, "reward_judge_quality_std": 0.1728696972131729, "reward_total_composite_mean": 0.5712888240814209, "reward_total_composite_std": 0.1664804220199585} {"timestamp_utc": "2026-04-13T09:41:17Z", "mode": "train", "global_step": 925, "epoch": 0.09291813159216473, "loss": 0.0564, "grad_norm": 13.910771369934082, "learning_rate": 7.2000000000000005e-06, "num_tokens": 1634440.0, "completions/mean_length": 48.375, "completions/min_length": 42.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.375, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.7714744806289673, "rewards/meter/std": 0.32325148582458496, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9265764951705933, "rewards/repeat_soft/std": 0.12223569303750992, "rewards/judge_quality/mean": 0.6599999666213989, "rewards/judge_quality/std": 0.2855571210384369, "rewards/total_composite/mean": 0.7035986185073853, "rewards/total_composite/std": 0.24538688361644745, "reward": 0.7035986185073853, "reward_std": 0.24538686871528625, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16183127462863922, "sampling/sampling_logp_difference/max": 1.7264628410339355, "sampling/importance_sampling_ratio/min": 0.17791259288787842, "sampling/importance_sampling_ratio/mean": 1.0172739028930664, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8879731148481369, "clip_ratio/low_mean": 0.08377291262149811, "clip_ratio/low_min": 0.08377291262149811, "clip_ratio/high_mean": 0.08953542355448008, "clip_ratio/high_max": 0.08953542355448008, "clip_ratio/region_mean": 0.17330833617597818, "reward_total_mean": 0.7035986185073853, "reward_meter_mean": 0.7714744806289673, "reward_meter_std": 0.32325148582458496, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9265764951705933, "reward_repeat_soft_std": 0.12223569303750992, "reward_judge_quality_mean": 0.6599999666213989, "reward_judge_quality_std": 0.2855571210384369, "reward_total_composite_mean": 0.7035986185073853, "reward_total_composite_std": 0.24538688361644745} {"timestamp_utc": "2026-04-13T09:41:24Z", "mode": "train", "global_step": 926, "epoch": 0.09301858362631843, "loss": 0.0661, "grad_norm": 7.966991901397705, "learning_rate": 7.196969696969698e-06, "num_tokens": 1637037.0, "completions/mean_length": 107.625, "completions/min_length": 98.0, "completions/max_length": 121.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.625, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.6245505809783936, "rewards/meter/std": 0.3284919559955597, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.05892555043101311, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9416639804840088, "rewards/repeat_soft/std": 0.04125182330608368, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.5026393532752991, "rewards/total_composite/std": 0.1356973648071289, "reward": 0.5026393532752991, "reward_std": 0.1356973648071289, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15753981471061707, "sampling/sampling_logp_difference/max": 1.847491979598999, "sampling/importance_sampling_ratio/min": 0.1576320081949234, "sampling/importance_sampling_ratio/mean": 1.0120426416397095, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9671919271349907, "clip_ratio/low_mean": 0.06904459372162819, "clip_ratio/low_min": 0.06904459372162819, "clip_ratio/high_mean": 0.05026556085795164, "clip_ratio/high_max": 0.05026556085795164, "clip_ratio/region_mean": 0.11931015457957983, "reward_total_mean": 0.5026393532752991, "reward_meter_mean": 0.6245505809783936, "reward_meter_std": 0.3284919559955597, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.05892555043101311, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9416639804840088, "reward_repeat_soft_std": 0.04125182330608368, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.5026393532752991, "reward_total_composite_std": 0.1356973648071289} {"timestamp_utc": "2026-04-13T09:41:30Z", "mode": "train", "global_step": 927, "epoch": 0.09311903566047212, "loss": 0.0252, "grad_norm": 9.569807052612305, "learning_rate": 7.193939393939394e-06, "num_tokens": 1639370.0, "completions/mean_length": 99.625, "completions/min_length": 91.0, "completions/max_length": 112.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 99.625, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 112.0, "rewards/meter/mean": 0.5098446011543274, "rewards/meter/std": 0.3082834482192993, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8003026247024536, "rewards/repeat_soft/std": 0.08848375082015991, "rewards/judge_quality/mean": 0.5325000286102295, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.5001412630081177, "rewards/total_composite/std": 0.13108134269714355, "reward": 0.5001412630081177, "reward_std": 0.13108134269714355, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13828051090240479, "sampling/sampling_logp_difference/max": 1.9262475967407227, "sampling/importance_sampling_ratio/min": 0.14569388329982758, "sampling/importance_sampling_ratio/mean": 0.9944371581077576, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7524937465786934, "clip_ratio/low_mean": 0.05981121212244034, "clip_ratio/low_min": 0.05981121212244034, "clip_ratio/high_mean": 0.07867600489407778, "clip_ratio/high_max": 0.07867600489407778, "clip_ratio/region_mean": 0.13848721701651812, "reward_total_mean": 0.5001412630081177, "reward_meter_mean": 0.5098446011543274, "reward_meter_std": 0.3082834482192993, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8003026247024536, "reward_repeat_soft_std": 0.08848375082015991, "reward_judge_quality_mean": 0.5325000286102295, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.5001412630081177, "reward_total_composite_std": 0.13108134269714355} {"timestamp_utc": "2026-04-13T09:41:36Z", "mode": "train", "global_step": 928, "epoch": 0.09321948769462582, "loss": 0.0807, "grad_norm": 17.609262466430664, "learning_rate": 7.1909090909090914e-06, "num_tokens": 1640986.0, "completions/mean_length": 38.0, "completions/min_length": 34.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.0, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.9794037342071533, "rewards/meter/std": 0.027005111798644066, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9568880796432495, "rewards/repeat_soft/std": 0.047611843794584274, "rewards/judge_quality/mean": 0.6737500429153442, "rewards/judge_quality/std": 0.263435423374176, "rewards/total_composite/mean": 0.7738229036331177, "rewards/total_composite/std": 0.172917902469635, "reward": 0.7738229036331177, "reward_std": 0.17291788756847382, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15360701084136963, "sampling/sampling_logp_difference/max": 1.5589860677719116, "sampling/importance_sampling_ratio/min": 0.21034923195838928, "sampling/importance_sampling_ratio/mean": 1.0201724767684937, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1102176755666733, "clip_ratio/low_mean": 0.08575666323304176, "clip_ratio/low_min": 0.08575666323304176, "clip_ratio/high_mean": 0.07010342739522457, "clip_ratio/high_max": 0.07010342739522457, "clip_ratio/region_mean": 0.15586009062826633, "reward_total_mean": 0.7738229036331177, "reward_meter_mean": 0.9794037342071533, "reward_meter_std": 0.027005111798644066, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9568880796432495, "reward_repeat_soft_std": 0.047611843794584274, "reward_judge_quality_mean": 0.6737500429153442, "reward_judge_quality_std": 0.263435423374176, "reward_total_composite_mean": 0.7738229036331177, "reward_total_composite_std": 0.172917902469635} {"timestamp_utc": "2026-04-13T09:41:47Z", "mode": "train", "global_step": 929, "epoch": 0.09331993972877951, "loss": -0.083, "grad_norm": 2.698056697845459, "learning_rate": 7.187878787878788e-06, "num_tokens": 1642428.0, "completions/mean_length": 86.25, "completions/min_length": 23.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 25.428571701049805, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.8588230609893799, "rewards/meter/std": 0.262309730052948, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9585226774215698, "rewards/repeat_soft/std": 0.011249415576457977, "rewards/judge_quality/mean": 0.38875001668930054, "rewards/judge_quality/std": 0.13767844438552856, "rewards/total_composite/mean": 0.5160880088806152, "rewards/total_composite/std": 0.22122929990291595, "reward": 0.5160880088806152, "reward_std": 0.22122929990291595, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16538824141025543, "sampling/sampling_logp_difference/max": 1.312190055847168, "sampling/importance_sampling_ratio/min": 0.26922979950904846, "sampling/importance_sampling_ratio/mean": 1.062637209892273, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2604689002037048, "clip_ratio/low_mean": 0.03125, "clip_ratio/low_min": 0.03125, "clip_ratio/high_mean": 0.09820446511730552, "clip_ratio/high_max": 0.09820446511730552, "clip_ratio/region_mean": 0.12945446511730552, "reward_total_mean": 0.5160880088806152, "reward_meter_mean": 0.8588230609893799, "reward_meter_std": 0.262309730052948, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9585226774215698, "reward_repeat_soft_std": 0.011249415576457977, "reward_judge_quality_mean": 0.38875001668930054, "reward_judge_quality_std": 0.13767844438552856, "reward_total_composite_mean": 0.5160880088806152, "reward_total_composite_std": 0.22122929990291595} {"timestamp_utc": "2026-04-13T09:41:53Z", "mode": "train", "global_step": 930, "epoch": 0.0934203917629332, "loss": -0.029, "grad_norm": 13.70984172821045, "learning_rate": 7.184848484848486e-06, "num_tokens": 1644120.0, "completions/mean_length": 46.5, "completions/min_length": 39.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.5, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.7361986041069031, "rewards/meter/std": 0.3028372526168823, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9762316942214966, "rewards/repeat_soft/std": 0.024908561259508133, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.5551185607910156, "rewards/total_composite/std": 0.08748355507850647, "reward": 0.5551185607910156, "reward_std": 0.08748354762792587, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1647631973028183, "sampling/sampling_logp_difference/max": 1.473271131515503, "sampling/importance_sampling_ratio/min": 0.28066200017929077, "sampling/importance_sampling_ratio/mean": 1.0438170433044434, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.39552640914917, "clip_ratio/low_mean": 0.06653808243572712, "clip_ratio/low_min": 0.06653808243572712, "clip_ratio/high_mean": 0.10128033626824617, "clip_ratio/high_max": 0.10128033626824617, "clip_ratio/region_mean": 0.1678184187039733, "reward_total_mean": 0.5551185607910156, "reward_meter_mean": 0.7361986041069031, "reward_meter_std": 0.3028372526168823, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9762316942214966, "reward_repeat_soft_std": 0.024908561259508133, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.5551185607910156, "reward_total_composite_std": 0.08748355507850647} {"timestamp_utc": "2026-04-13T09:42:06Z", "mode": "train", "global_step": 931, "epoch": 0.09352084379708689, "loss": 0.0733, "grad_norm": 20.699567794799805, "learning_rate": 7.181818181818182e-06, "num_tokens": 1645514.0, "completions/mean_length": 25.25, "completions/min_length": 23.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.25, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.3937675952911377, "rewards/meter/std": 0.2996700704097748, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9385433197021484, "rewards/repeat_soft/std": 0.04496556892991066, "rewards/judge_quality/mean": 0.7400000095367432, "rewards/judge_quality/std": 0.24859607219696045, "rewards/total_composite/mean": 0.5503440499305725, "rewards/total_composite/std": 0.18855828046798706, "reward": 0.5503440499305725, "reward_std": 0.18855829536914825, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17943686246871948, "sampling/sampling_logp_difference/max": 2.067169666290283, "sampling/importance_sampling_ratio/min": 0.12654343247413635, "sampling/importance_sampling_ratio/mean": 0.9980342984199524, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7266267389059067, "clip_ratio/low_mean": 0.09316770359873772, "clip_ratio/low_min": 0.09316770359873772, "clip_ratio/high_mean": 0.05708333197981119, "clip_ratio/high_max": 0.05708333197981119, "clip_ratio/region_mean": 0.1502510355785489, "reward_total_mean": 0.5503440499305725, "reward_meter_mean": 0.3937675952911377, "reward_meter_std": 0.2996700704097748, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9385433197021484, "reward_repeat_soft_std": 0.04496556892991066, "reward_judge_quality_mean": 0.7400000095367432, "reward_judge_quality_std": 0.24859607219696045, "reward_total_composite_mean": 0.5503440499305725, "reward_total_composite_std": 0.18855828046798706} {"timestamp_utc": "2026-04-13T09:42:12Z", "mode": "train", "global_step": 932, "epoch": 0.09362129583124058, "loss": 0.0617, "grad_norm": 9.303230285644531, "learning_rate": 7.17878787878788e-06, "num_tokens": 1647840.0, "completions/mean_length": 97.75, "completions/min_length": 91.0, "completions/max_length": 110.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.75, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 110.0, "rewards/meter/mean": 0.9708800315856934, "rewards/meter/std": 0.014429431408643723, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.855011522769928, "rewards/repeat_soft/std": 0.09349798411130905, "rewards/judge_quality/mean": 0.39375001192092896, "rewards/judge_quality/std": 0.15638209879398346, "rewards/total_composite/mean": 0.535617470741272, "rewards/total_composite/std": 0.09823641926050186, "reward": 0.535617470741272, "reward_std": 0.09823642671108246, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14550121128559113, "sampling/sampling_logp_difference/max": 2.029776096343994, "sampling/importance_sampling_ratio/min": 0.13136492669582367, "sampling/importance_sampling_ratio/mean": 1.020607352256775, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0373662635684013, "clip_ratio/low_mean": 0.045313782058656216, "clip_ratio/low_min": 0.045313782058656216, "clip_ratio/high_mean": 0.08563645090907812, "clip_ratio/high_max": 0.08563645090907812, "clip_ratio/region_mean": 0.13095023296773434, "reward_total_mean": 0.535617470741272, "reward_meter_mean": 0.9708800315856934, "reward_meter_std": 0.014429431408643723, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.855011522769928, "reward_repeat_soft_std": 0.09349798411130905, "reward_judge_quality_mean": 0.39375001192092896, "reward_judge_quality_std": 0.15638209879398346, "reward_total_composite_mean": 0.535617470741272, "reward_total_composite_std": 0.09823641926050186} {"timestamp_utc": "2026-04-13T09:42:20Z", "mode": "train", "global_step": 933, "epoch": 0.09372174786539428, "loss": 0.0276, "grad_norm": 25.0538330078125, "learning_rate": 7.175757575757576e-06, "num_tokens": 1649272.0, "completions/mean_length": 23.0, "completions/min_length": 18.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.0, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.8249565958976746, "rewards/meter/std": 0.32491472363471985, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9525578022003174, "rewards/repeat_soft/std": 0.014416005462408066, "rewards/judge_quality/mean": 0.39750000834465027, "rewards/judge_quality/std": 0.10110107809305191, "rewards/total_composite/mean": 0.5508995056152344, "rewards/total_composite/std": 0.09900034964084625, "reward": 0.5508995056152344, "reward_std": 0.09900033473968506, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17802026867866516, "sampling/sampling_logp_difference/max": 1.075533390045166, "sampling/importance_sampling_ratio/min": 0.3726882338523865, "sampling/importance_sampling_ratio/mean": 1.0378916263580322, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3319678083062172, "clip_ratio/low_mean": 0.06595849804580212, "clip_ratio/low_min": 0.06595849804580212, "clip_ratio/high_mean": 0.09105158830061555, "clip_ratio/high_max": 0.09105158830061555, "clip_ratio/region_mean": 0.15701008634641767, "reward_total_mean": 0.5508995056152344, "reward_meter_mean": 0.8249565958976746, "reward_meter_std": 0.32491472363471985, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9525578022003174, "reward_repeat_soft_std": 0.014416005462408066, "reward_judge_quality_mean": 0.39750000834465027, "reward_judge_quality_std": 0.10110107809305191, "reward_total_composite_mean": 0.5508995056152344, "reward_total_composite_std": 0.09900034964084625} {"timestamp_utc": "2026-04-13T09:42:26Z", "mode": "train", "global_step": 934, "epoch": 0.09382219989954796, "loss": 0.0493, "grad_norm": 15.443513870239258, "learning_rate": 7.172727272727273e-06, "num_tokens": 1651053.0, "completions/mean_length": 48.625, "completions/min_length": 45.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.625, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.7085595726966858, "rewards/meter/std": 0.3993377387523651, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9652116298675537, "rewards/repeat_soft/std": 0.03574341535568237, "rewards/judge_quality/mean": 0.6112499833106995, "rewards/judge_quality/std": 0.15037456154823303, "rewards/total_composite/mean": 0.6232345104217529, "rewards/total_composite/std": 0.1727263331413269, "reward": 0.6232345104217529, "reward_std": 0.1727263331413269, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1562386006116867, "sampling/sampling_logp_difference/max": 1.3455955982208252, "sampling/importance_sampling_ratio/min": 0.26038458943367004, "sampling/importance_sampling_ratio/mean": 1.0110880136489868, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1191803216934204, "clip_ratio/low_mean": 0.09817763511091471, "clip_ratio/low_min": 0.09817763511091471, "clip_ratio/high_mean": 0.061666665598750114, "clip_ratio/high_max": 0.061666665598750114, "clip_ratio/region_mean": 0.15984430070966482, "reward_total_mean": 0.6232345104217529, "reward_meter_mean": 0.7085595726966858, "reward_meter_std": 0.3993377387523651, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9652116298675537, "reward_repeat_soft_std": 0.03574341535568237, "reward_judge_quality_mean": 0.6112499833106995, "reward_judge_quality_std": 0.15037456154823303, "reward_total_composite_mean": 0.6232345104217529, "reward_total_composite_std": 0.1727263331413269} {"timestamp_utc": "2026-04-13T09:42:32Z", "mode": "train", "global_step": 935, "epoch": 0.09392265193370165, "loss": 0.0255, "grad_norm": 24.60488510131836, "learning_rate": 7.16969696969697e-06, "num_tokens": 1652549.0, "completions/mean_length": 21.0, "completions/min_length": 19.0, "completions/max_length": 23.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.0, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 23.0, "rewards/meter/mean": 0.9167644381523132, "rewards/meter/std": 0.07410956174135208, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9464983940124512, "rewards/repeat_soft/std": 0.04525930806994438, "rewards/judge_quality/mean": 0.4137499928474426, "rewards/judge_quality/std": 0.06781013309955597, "rewards/total_composite/mean": 0.586983323097229, "rewards/total_composite/std": 0.04325531795620918, "reward": 0.586983323097229, "reward_std": 0.04325530305504799, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14468401670455933, "sampling/sampling_logp_difference/max": 1.8507564067840576, "sampling/importance_sampling_ratio/min": 0.1571182757616043, "sampling/importance_sampling_ratio/mean": 1.015199899673462, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9049945846199989, "clip_ratio/low_mean": 0.04805492050945759, "clip_ratio/low_min": 0.04805492050945759, "clip_ratio/high_mean": 0.06471861619502306, "clip_ratio/high_max": 0.06471861619502306, "clip_ratio/region_mean": 0.11277353670448065, "reward_total_mean": 0.586983323097229, "reward_meter_mean": 0.9167644381523132, "reward_meter_std": 0.07410956174135208, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9464983940124512, "reward_repeat_soft_std": 0.04525930806994438, "reward_judge_quality_mean": 0.4137499928474426, "reward_judge_quality_std": 0.06781013309955597, "reward_total_composite_mean": 0.586983323097229, "reward_total_composite_std": 0.04325531795620918} {"timestamp_utc": "2026-04-13T09:42:38Z", "mode": "train", "global_step": 936, "epoch": 0.09402310396785535, "loss": 0.0007, "grad_norm": 11.984716415405273, "learning_rate": 7.166666666666667e-06, "num_tokens": 1654466.0, "completions/mean_length": 62.625, "completions/min_length": 56.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.625, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9602702856063843, "rewards/meter/std": 0.019833838567137718, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9538745880126953, "rewards/repeat_soft/std": 0.014454740099608898, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.5918318629264832, "rewards/total_composite/std": 0.03716662898659706, "reward": 0.5918318629264832, "reward_std": 0.03716662526130676, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15173695981502533, "sampling/sampling_logp_difference/max": 2.329094409942627, "sampling/importance_sampling_ratio/min": 0.09738390147686005, "sampling/importance_sampling_ratio/mean": 1.0248721837997437, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1424496248364449, "clip_ratio/low_mean": 0.0366459209471941, "clip_ratio/low_min": 0.0366459209471941, "clip_ratio/high_mean": 0.12486749235540628, "clip_ratio/high_max": 0.12486749235540628, "clip_ratio/region_mean": 0.16151341330260038, "reward_total_mean": 0.5918318629264832, "reward_meter_mean": 0.9602702856063843, "reward_meter_std": 0.019833838567137718, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9538745880126953, "reward_repeat_soft_std": 0.014454740099608898, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.5918318629264832, "reward_total_composite_std": 0.03716662898659706} {"timestamp_utc": "2026-04-13T09:42:45Z", "mode": "train", "global_step": 937, "epoch": 0.09412355600200904, "loss": 0.012, "grad_norm": 11.2230863571167, "learning_rate": 7.163636363636363e-06, "num_tokens": 1656597.0, "completions/mean_length": 96.375, "completions/min_length": 91.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.375, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.10800344496965408, "rewards/meter/std": 0.11452758312225342, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8984007835388184, "rewards/repeat_soft/std": 0.12824523448944092, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.3662221133708954, "rewards/total_composite/std": 0.040435004979372025, "reward": 0.3662221133708954, "reward_std": 0.04043500870466232, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1392218917608261, "sampling/sampling_logp_difference/max": 1.3230323791503906, "sampling/importance_sampling_ratio/min": 0.2663264870643616, "sampling/importance_sampling_ratio/mean": 1.0213192701339722, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9289983585476875, "clip_ratio/low_mean": 0.10917292069643736, "clip_ratio/low_min": 0.10917292069643736, "clip_ratio/high_mean": 0.031905731186270714, "clip_ratio/high_max": 0.031905731186270714, "clip_ratio/region_mean": 0.14107865188270807, "reward_total_mean": 0.3662221133708954, "reward_meter_mean": 0.10800344496965408, "reward_meter_std": 0.11452758312225342, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8984007835388184, "reward_repeat_soft_std": 0.12824523448944092, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.3662221133708954, "reward_total_composite_std": 0.040435004979372025} {"timestamp_utc": "2026-04-13T09:42:51Z", "mode": "train", "global_step": 938, "epoch": 0.09422400803616274, "loss": 0.0011, "grad_norm": 9.277302742004395, "learning_rate": 7.1606060606060615e-06, "num_tokens": 1658376.0, "completions/mean_length": 63.375, "completions/min_length": 55.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.375, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.8871058225631714, "rewards/meter/std": 0.12349243462085724, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9183275699615479, "rewards/repeat_soft/std": 0.05043455958366394, "rewards/judge_quality/mean": 0.47749996185302734, "rewards/judge_quality/std": 0.16263456642627716, "rewards/total_composite/mean": 0.6136891841888428, "rewards/total_composite/std": 0.1065220907330513, "reward": 0.6136891841888428, "reward_std": 0.1065220981836319, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1586700975894928, "sampling/sampling_logp_difference/max": 1.6482105255126953, "sampling/importance_sampling_ratio/min": 0.19239388406276703, "sampling/importance_sampling_ratio/mean": 1.0129984617233276, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0819894894957542, "clip_ratio/low_mean": 0.12331399507820606, "clip_ratio/low_min": 0.12331399507820606, "clip_ratio/high_mean": 0.014492753893136978, "clip_ratio/high_max": 0.014492753893136978, "clip_ratio/region_mean": 0.13780674897134304, "reward_total_mean": 0.6136891841888428, "reward_meter_mean": 0.8871058225631714, "reward_meter_std": 0.12349243462085724, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9183275699615479, "reward_repeat_soft_std": 0.05043455958366394, "reward_judge_quality_mean": 0.47749996185302734, "reward_judge_quality_std": 0.16263456642627716, "reward_total_composite_mean": 0.6136891841888428, "reward_total_composite_std": 0.1065220907330513} {"timestamp_utc": "2026-04-13T09:42:57Z", "mode": "train", "global_step": 939, "epoch": 0.09432446007031642, "loss": 0.0903, "grad_norm": 14.62889289855957, "learning_rate": 7.157575757575758e-06, "num_tokens": 1660098.0, "completions/mean_length": 47.25, "completions/min_length": 40.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.25, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.7967574596405029, "rewards/meter/std": 0.32078492641448975, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9507006406784058, "rewards/repeat_soft/std": 0.043021660298109055, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.566234827041626, "rewards/total_composite/std": 0.08880805224180222, "reward": 0.566234827041626, "reward_std": 0.08880805224180222, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14013466238975525, "sampling/sampling_logp_difference/max": 1.2640771865844727, "sampling/importance_sampling_ratio/min": 0.3426993191242218, "sampling/importance_sampling_ratio/mean": 1.023146629333496, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.096999168395996, "clip_ratio/low_mean": 0.034519231878221035, "clip_ratio/low_min": 0.034519231878221035, "clip_ratio/high_mean": 0.14281363226473331, "clip_ratio/high_max": 0.14281363226473331, "clip_ratio/region_mean": 0.17733286414295435, "reward_total_mean": 0.566234827041626, "reward_meter_mean": 0.7967574596405029, "reward_meter_std": 0.32078492641448975, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9507006406784058, "reward_repeat_soft_std": 0.043021660298109055, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.566234827041626, "reward_total_composite_std": 0.08880805224180222} {"timestamp_utc": "2026-04-13T09:43:03Z", "mode": "train", "global_step": 940, "epoch": 0.09442491210447011, "loss": -0.025, "grad_norm": 11.421670913696289, "learning_rate": 7.154545454545455e-06, "num_tokens": 1661672.0, "completions/mean_length": 45.75, "completions/min_length": 40.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.75, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.5459588766098022, "rewards/meter/std": 0.4084034562110901, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9860960245132446, "rewards/repeat_soft/std": 0.015939654782414436, "rewards/judge_quality/mean": 0.6775000095367432, "rewards/judge_quality/std": 0.25949129462242126, "rewards/total_composite/mean": 0.5541008114814758, "rewards/total_composite/std": 0.17735260725021362, "reward": 0.5541008114814758, "reward_std": 0.17735260725021362, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13485819101333618, "sampling/sampling_logp_difference/max": 1.647550106048584, "sampling/importance_sampling_ratio/min": 0.19252097606658936, "sampling/importance_sampling_ratio/mean": 0.9911300539970398, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7478036507964134, "clip_ratio/low_mean": 0.08339780941605568, "clip_ratio/low_min": 0.08339780941605568, "clip_ratio/high_mean": 0.05372035503387451, "clip_ratio/high_max": 0.05372035503387451, "clip_ratio/region_mean": 0.1371181644499302, "reward_total_mean": 0.5541008114814758, "reward_meter_mean": 0.5459588766098022, "reward_meter_std": 0.4084034562110901, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9860960245132446, "reward_repeat_soft_std": 0.015939654782414436, "reward_judge_quality_mean": 0.6775000095367432, "reward_judge_quality_std": 0.25949129462242126, "reward_total_composite_mean": 0.5541008114814758, "reward_total_composite_std": 0.17735260725021362} {"timestamp_utc": "2026-04-13T09:43:14Z", "mode": "train", "global_step": 941, "epoch": 0.09452536413862381, "loss": -0.069, "grad_norm": 2.9967703819274902, "learning_rate": 7.151515151515152e-06, "num_tokens": 1662933.0, "completions/mean_length": 85.625, "completions/min_length": 22.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 24.71428680419922, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.6023797988891602, "rewards/meter/std": 0.36645641922950745, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9555096626281738, "rewards/repeat_soft/std": 0.01863044500350952, "rewards/judge_quality/mean": 0.36000001430511475, "rewards/judge_quality/std": 0.140813946723938, "rewards/total_composite/mean": 0.4410613775253296, "rewards/total_composite/std": 0.20302338898181915, "reward": 0.4410613775253296, "reward_std": 0.20302337408065796, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1412070393562317, "sampling/sampling_logp_difference/max": 0.9118883609771729, "sampling/importance_sampling_ratio/min": 0.401764839887619, "sampling/importance_sampling_ratio/mean": 0.9974734783172607, "sampling/importance_sampling_ratio/max": 1.8005191087722778, "entropy": 0.8800483867526054, "clip_ratio/low_mean": 0.04326923284679651, "clip_ratio/low_min": 0.04326923284679651, "clip_ratio/high_mean": 0.08372922521084547, "clip_ratio/high_max": 0.08372922521084547, "clip_ratio/region_mean": 0.12699845805764198, "reward_total_mean": 0.4410613775253296, "reward_meter_mean": 0.6023797988891602, "reward_meter_std": 0.36645641922950745, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9555096626281738, "reward_repeat_soft_std": 0.01863044500350952, "reward_judge_quality_mean": 0.36000001430511475, "reward_judge_quality_std": 0.140813946723938, "reward_total_composite_mean": 0.4410613775253296, "reward_total_composite_std": 0.20302338898181915} {"timestamp_utc": "2026-04-13T09:43:20Z", "mode": "train", "global_step": 942, "epoch": 0.0946258161727775, "loss": 0.0174, "grad_norm": 18.894412994384766, "learning_rate": 7.148484848484849e-06, "num_tokens": 1664417.0, "completions/mean_length": 23.5, "completions/min_length": 19.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.5, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.6397724747657776, "rewards/meter/std": 0.4632761776447296, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9264246225357056, "rewards/repeat_soft/std": 0.03450237587094307, "rewards/judge_quality/mean": 0.6525000333786011, "rewards/judge_quality/std": 0.2921227812767029, "rewards/total_composite/mean": 0.5938670635223389, "rewards/total_composite/std": 0.23742662370204926, "reward": 0.5938670635223389, "reward_std": 0.23742660880088806, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16714201867580414, "sampling/sampling_logp_difference/max": 1.317133903503418, "sampling/importance_sampling_ratio/min": 0.2679020166397095, "sampling/importance_sampling_ratio/mean": 1.0264168977737427, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0534771531820297, "clip_ratio/low_mean": 0.09427609574049711, "clip_ratio/low_min": 0.09427609574049711, "clip_ratio/high_mean": 0.05899122916162014, "clip_ratio/high_max": 0.05899122916162014, "clip_ratio/region_mean": 0.15326732490211725, "reward_total_mean": 0.5938670635223389, "reward_meter_mean": 0.6397724747657776, "reward_meter_std": 0.4632761776447296, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9264246225357056, "reward_repeat_soft_std": 0.03450237587094307, "reward_judge_quality_mean": 0.6525000333786011, "reward_judge_quality_std": 0.2921227812767029, "reward_total_composite_mean": 0.5938670635223389, "reward_total_composite_std": 0.23742662370204926} {"timestamp_utc": "2026-04-13T09:43:27Z", "mode": "train", "global_step": 943, "epoch": 0.0947262682069312, "loss": 0.0319, "grad_norm": 11.56407356262207, "learning_rate": 7.145454545454547e-06, "num_tokens": 1666637.0, "completions/mean_length": 73.5, "completions/min_length": 68.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.5, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.7723885774612427, "rewards/meter/std": 0.23276889324188232, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9819412231445312, "rewards/repeat_soft/std": 0.020347705110907555, "rewards/judge_quality/mean": 0.6950000524520874, "rewards/judge_quality/std": 0.1908627301454544, "rewards/total_composite/mean": 0.6922646760940552, "rewards/total_composite/std": 0.12439776957035065, "reward": 0.6922646760940552, "reward_std": 0.12439776957035065, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12646839022636414, "sampling/sampling_logp_difference/max": 1.4681923389434814, "sampling/importance_sampling_ratio/min": 0.23034149408340454, "sampling/importance_sampling_ratio/mean": 1.0263735055923462, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7399717159569263, "clip_ratio/low_mean": 0.06865095626562834, "clip_ratio/low_min": 0.06865095626562834, "clip_ratio/high_mean": 0.05612659826874733, "clip_ratio/high_max": 0.05612659826874733, "clip_ratio/region_mean": 0.12477755453437567, "reward_total_mean": 0.6922646760940552, "reward_meter_mean": 0.7723885774612427, "reward_meter_std": 0.23276889324188232, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9819412231445312, "reward_repeat_soft_std": 0.020347705110907555, "reward_judge_quality_mean": 0.6950000524520874, "reward_judge_quality_std": 0.1908627301454544, "reward_total_composite_mean": 0.6922646760940552, "reward_total_composite_std": 0.12439776957035065} {"timestamp_utc": "2026-04-13T09:43:38Z", "mode": "train", "global_step": 944, "epoch": 0.09482672024108488, "loss": -0.0716, "grad_norm": 2.4013564586639404, "learning_rate": 7.142424242424243e-06, "num_tokens": 1668010.0, "completions/mean_length": 83.625, "completions/min_length": 17.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 22.428571701049805, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.6109882593154907, "rewards/meter/std": 0.44488978385925293, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9552083015441895, "rewards/repeat_soft/std": 0.02062394842505455, "rewards/judge_quality/mean": 0.2462500035762787, "rewards/judge_quality/std": 0.09913014620542526, "rewards/total_composite/mean": 0.41316545009613037, "rewards/total_composite/std": 0.18364334106445312, "reward": 0.41316545009613037, "reward_std": 0.18364332616329193, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1917419284582138, "sampling/sampling_logp_difference/max": 1.7236121892929077, "sampling/importance_sampling_ratio/min": 0.17842049896717072, "sampling/importance_sampling_ratio/mean": 0.9943436980247498, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0650127157568932, "clip_ratio/low_mean": 0.03967391233891249, "clip_ratio/low_min": 0.03967391233891249, "clip_ratio/high_mean": 0.11422209721058607, "clip_ratio/high_max": 0.11422209721058607, "clip_ratio/region_mean": 0.15389600954949856, "reward_total_mean": 0.41316545009613037, "reward_meter_mean": 0.6109882593154907, "reward_meter_std": 0.44488978385925293, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9552083015441895, "reward_repeat_soft_std": 0.02062394842505455, "reward_judge_quality_mean": 0.2462500035762787, "reward_judge_quality_std": 0.09913014620542526, "reward_total_composite_mean": 0.41316545009613037, "reward_total_composite_std": 0.18364334106445312} {"timestamp_utc": "2026-04-13T09:43:44Z", "mode": "train", "global_step": 945, "epoch": 0.09492717227523857, "loss": -0.0008, "grad_norm": 11.407129287719727, "learning_rate": 7.1393939393939405e-06, "num_tokens": 1669694.0, "completions/mean_length": 55.5, "completions/min_length": 39.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.5, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.4709435701370239, "rewards/meter/std": 0.409063458442688, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9676225185394287, "rewards/repeat_soft/std": 0.028310758993029594, "rewards/judge_quality/mean": 0.46875, "rewards/judge_quality/std": 0.10260014235973358, "rewards/total_composite/mean": 0.47954416275024414, "rewards/total_composite/std": 0.11127520352602005, "reward": 0.47954416275024414, "reward_std": 0.11127523332834244, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16715635359287262, "sampling/sampling_logp_difference/max": 1.3540549278259277, "sampling/importance_sampling_ratio/min": 0.25819119811058044, "sampling/importance_sampling_ratio/mean": 1.025413990020752, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4451678767800331, "clip_ratio/low_mean": 0.11211142037063837, "clip_ratio/low_min": 0.11211142037063837, "clip_ratio/high_mean": 0.059318218380212784, "clip_ratio/high_max": 0.059318218380212784, "clip_ratio/region_mean": 0.17142963875085115, "reward_total_mean": 0.47954416275024414, "reward_meter_mean": 0.4709435701370239, "reward_meter_std": 0.409063458442688, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9676225185394287, "reward_repeat_soft_std": 0.028310758993029594, "reward_judge_quality_mean": 0.46875, "reward_judge_quality_std": 0.10260014235973358, "reward_total_composite_mean": 0.47954416275024414, "reward_total_composite_std": 0.11127520352602005} {"timestamp_utc": "2026-04-13T09:43:51Z", "mode": "train", "global_step": 946, "epoch": 0.09502762430939227, "loss": 0.0241, "grad_norm": 9.67296314239502, "learning_rate": 7.136363636363637e-06, "num_tokens": 1672116.0, "completions/mean_length": 103.75, "completions/min_length": 94.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 103.75, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.955755889415741, "rewards/meter/std": 0.0636955127120018, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8535679578781128, "rewards/repeat_soft/std": 0.04481218382716179, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.5544466972351074, "rewards/total_composite/std": 0.07041594386100769, "reward": 0.5544466972351074, "reward_std": 0.07041595131158829, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13135242462158203, "sampling/sampling_logp_difference/max": 2.8089582920074463, "sampling/importance_sampling_ratio/min": 0.06026774272322655, "sampling/importance_sampling_ratio/mean": 1.0094482898712158, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9232650846242905, "clip_ratio/low_mean": 0.0405056644231081, "clip_ratio/low_min": 0.0405056644231081, "clip_ratio/high_mean": 0.09079292230308056, "clip_ratio/high_max": 0.09079292230308056, "clip_ratio/region_mean": 0.13129858672618866, "reward_total_mean": 0.5544466972351074, "reward_meter_mean": 0.955755889415741, "reward_meter_std": 0.0636955127120018, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8535679578781128, "reward_repeat_soft_std": 0.04481218382716179, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.5544466972351074, "reward_total_composite_std": 0.07041594386100769} {"timestamp_utc": "2026-04-13T09:43:58Z", "mode": "train", "global_step": 947, "epoch": 0.09512807634354596, "loss": -0.0478, "grad_norm": 10.617412567138672, "learning_rate": 7.133333333333334e-06, "num_tokens": 1674288.0, "completions/mean_length": 87.5, "completions/min_length": 74.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.5, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.5281727313995361, "rewards/meter/std": 0.2483045905828476, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9489359855651855, "rewards/repeat_soft/std": 0.024047967046499252, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.48799845576286316, "rewards/total_composite/std": 0.06793003529310226, "reward": 0.48799845576286316, "reward_std": 0.06793004274368286, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15323258936405182, "sampling/sampling_logp_difference/max": 2.0040841102600098, "sampling/importance_sampling_ratio/min": 0.13478368520736694, "sampling/importance_sampling_ratio/mean": 0.9981544017791748, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9498008489608765, "clip_ratio/low_mean": 0.049493699334561825, "clip_ratio/low_min": 0.049493699334561825, "clip_ratio/high_mean": 0.09747860673815012, "clip_ratio/high_max": 0.09747860673815012, "clip_ratio/region_mean": 0.14697230607271194, "reward_total_mean": 0.48799845576286316, "reward_meter_mean": 0.5281727313995361, "reward_meter_std": 0.2483045905828476, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9489359855651855, "reward_repeat_soft_std": 0.024047967046499252, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.48799845576286316, "reward_total_composite_std": 0.06793003529310226} {"timestamp_utc": "2026-04-13T09:44:09Z", "mode": "train", "global_step": 948, "epoch": 0.09522852837769964, "loss": -0.0637, "grad_norm": 1.6304961442947388, "learning_rate": 7.130303030303031e-06, "num_tokens": 1675619.0, "completions/mean_length": 147.375, "completions/min_length": 25.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 25.83333396911621, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.6447250247001648, "rewards/meter/std": 0.39510872960090637, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9657489061355591, "rewards/repeat_soft/std": 0.014413577504456043, "rewards/judge_quality/mean": 0.3462499976158142, "rewards/judge_quality/std": 0.18314221501350403, "rewards/total_composite/mean": 0.4207678437232971, "rewards/total_composite/std": 0.27831757068634033, "reward": 0.4207678437232971, "reward_std": 0.27831754088401794, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13598209619522095, "sampling/sampling_logp_difference/max": 1.261570930480957, "sampling/importance_sampling_ratio/min": 0.28320878744125366, "sampling/importance_sampling_ratio/mean": 1.0316158533096313, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8389666080474854, "clip_ratio/low_mean": 0.004999999888241291, "clip_ratio/low_min": 0.004999999888241291, "clip_ratio/high_mean": 0.08720085490494967, "clip_ratio/high_max": 0.08720085490494967, "clip_ratio/region_mean": 0.09220085479319096, "reward_total_mean": 0.4207678437232971, "reward_meter_mean": 0.6447250247001648, "reward_meter_std": 0.39510872960090637, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9657489061355591, "reward_repeat_soft_std": 0.014413577504456043, "reward_judge_quality_mean": 0.3462499976158142, "reward_judge_quality_std": 0.18314221501350403, "reward_total_composite_mean": 0.4207678437232971, "reward_total_composite_std": 0.27831757068634033} {"timestamp_utc": "2026-04-13T09:44:14Z", "mode": "train", "global_step": 949, "epoch": 0.09532898041185334, "loss": -0.0106, "grad_norm": 21.439594268798828, "learning_rate": 7.127272727272728e-06, "num_tokens": 1676994.0, "completions/mean_length": 23.875, "completions/min_length": 16.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.875, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9674949645996094, "rewards/meter/std": 0.03753045201301575, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9585226774215698, "rewards/repeat_soft/std": 0.011249415576457977, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.22403763234615326, "rewards/total_composite/mean": 0.6150607466697693, "rewards/total_composite/std": 0.14532624185085297, "reward": 0.6150607466697693, "reward_std": 0.14532622694969177, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14376404881477356, "sampling/sampling_logp_difference/max": 1.6640040874481201, "sampling/importance_sampling_ratio/min": 0.39237943291664124, "sampling/importance_sampling_ratio/mean": 1.0260133743286133, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9445859491825104, "clip_ratio/low_mean": 0.049259479623287916, "clip_ratio/low_min": 0.049259479623287916, "clip_ratio/high_mean": 0.01993534481152892, "clip_ratio/high_max": 0.01993534481152892, "clip_ratio/region_mean": 0.06919482443481684, "reward_total_mean": 0.6150607466697693, "reward_meter_mean": 0.9674949645996094, "reward_meter_std": 0.03753045201301575, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9585226774215698, "reward_repeat_soft_std": 0.011249415576457977, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.22403763234615326, "reward_total_composite_mean": 0.6150607466697693, "reward_total_composite_std": 0.14532624185085297} {"timestamp_utc": "2026-04-13T09:44:21Z", "mode": "train", "global_step": 950, "epoch": 0.09542943244600703, "loss": 0.0616, "grad_norm": 10.279980659484863, "learning_rate": 7.124242424242424e-06, "num_tokens": 1678891.0, "completions/mean_length": 73.125, "completions/min_length": 65.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.125, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.864984393119812, "rewards/meter/std": 0.13354484736919403, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8937699794769287, "rewards/repeat_soft/std": 0.06704690307378769, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.5357272028923035, "rewards/total_composite/std": 0.06531858444213867, "reward": 0.5357272028923035, "reward_std": 0.06531858444213867, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1296449452638626, "sampling/sampling_logp_difference/max": 1.644510269165039, "sampling/importance_sampling_ratio/min": 0.19310709834098816, "sampling/importance_sampling_ratio/mean": 1.0150951147079468, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7161218076944351, "clip_ratio/low_mean": 0.03791075386106968, "clip_ratio/low_min": 0.03791075386106968, "clip_ratio/high_mean": 0.06450229743495584, "clip_ratio/high_max": 0.06450229743495584, "clip_ratio/region_mean": 0.10241305129602551, "reward_total_mean": 0.5357272028923035, "reward_meter_mean": 0.864984393119812, "reward_meter_std": 0.13354484736919403, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8937699794769287, "reward_repeat_soft_std": 0.06704690307378769, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.5357272028923035, "reward_total_composite_std": 0.06531858444213867} {"timestamp_utc": "2026-04-13T09:45:11Z", "mode": "eval", "global_step": 950, "epoch": 0.09542943244600703, "eval_loss": NaN, "eval_runtime": 50.8034, "eval_samples_per_second": 1.575, "eval_steps_per_second": 0.197, "eval_num_tokens": 1678891.0, "eval_completions/mean_length": 82.2, "eval_completions/min_length": 32.6, "eval_completions/max_length": 196.9, "eval_completions/clipped_ratio": 0.025, "eval_completions/mean_terminated_length": 71.36071472167968, "eval_completions/min_terminated_length": 32.6, "eval_completions/max_terminated_length": 117.7, "eval_rewards/meter/mean": 0.6677819490432739, "eval_rewards/meter/std": 0.3473196476697922, "eval_rewards/count_adherence/mean": 0.9433333158493042, "eval_rewards/count_adherence/std": 0.0914482433348894, "eval_rewards/hard_gate/mean": 0.9375, "eval_rewards/hard_gate/std": 0.1767766922712326, "eval_rewards/repeat_soft/mean": 0.950187748670578, "eval_rewards/repeat_soft/std": 0.0460129925981164, "eval_rewards/judge_quality/mean": 0.4873750001192093, "eval_rewards/judge_quality/std": 0.20080990493297576, "eval_rewards/total_composite/mean": 0.5184092968702316, "eval_rewards/total_composite/std": 0.18421085253357888, "eval_reward": 0.5184092968702316, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.08459883593022824, "eval_sampling/sampling_logp_difference/max": 1.1512426376342773, "eval_sampling/importance_sampling_ratio/min": 0.32473981827497483, "eval_sampling/importance_sampling_ratio/mean": 1.0215644121170044, "eval_sampling/importance_sampling_ratio/max": 1.4665158629417419, "eval_entropy": 0.9833505392074585, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5184092968702316, "eval_reward_meter_mean": 0.6677819490432739, "eval_reward_meter_std": 0.3473196476697922, "eval_reward_count_adherence_mean": 0.9433333158493042, "eval_reward_count_adherence_std": 0.0914482433348894, "eval_reward_hard_gate_mean": 0.9375, "eval_reward_hard_gate_std": 0.1767766922712326, "eval_reward_repeat_soft_mean": 0.950187748670578, "eval_reward_repeat_soft_std": 0.0460129925981164, "eval_reward_judge_quality_mean": 0.4873750001192093, "eval_reward_judge_quality_std": 0.20080990493297576, "eval_reward_total_composite_mean": 0.5184092968702316, "eval_reward_total_composite_std": 0.18421085253357888} {"timestamp_utc": "2026-04-13T09:45:22Z", "mode": "train", "global_step": 951, "epoch": 0.09552988448016073, "loss": 0.0392, "grad_norm": 12.941825866699219, "learning_rate": 7.121212121212122e-06, "num_tokens": 1680537.0, "completions/mean_length": 50.75, "completions/min_length": 46.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.75, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.8829889297485352, "rewards/meter/std": 0.169756680727005, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9108825922012329, "rewards/repeat_soft/std": 0.09247001260519028, "rewards/judge_quality/mean": 0.6449999809265137, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.7023867964744568, "rewards/total_composite/std": 0.09584323316812515, "reward": 0.7023867964744568, "reward_std": 0.09584324061870575, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15451835095882416, "sampling/sampling_logp_difference/max": 1.7626707553863525, "sampling/importance_sampling_ratio/min": 0.17158597707748413, "sampling/importance_sampling_ratio/mean": 0.9980716705322266, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.186119705438614, "clip_ratio/low_mean": 0.05622141156345606, "clip_ratio/low_min": 0.05622141156345606, "clip_ratio/high_mean": 0.08247832767665386, "clip_ratio/high_max": 0.08247832767665386, "clip_ratio/region_mean": 0.13869973924010992, "reward_total_mean": 0.7023867964744568, "reward_meter_mean": 0.8829889297485352, "reward_meter_std": 0.169756680727005, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9108825922012329, "reward_repeat_soft_std": 0.09247001260519028, "reward_judge_quality_mean": 0.6449999809265137, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.7023867964744568, "reward_total_composite_std": 0.09584323316812515} {"timestamp_utc": "2026-04-13T09:45:29Z", "mode": "train", "global_step": 952, "epoch": 0.09563033651431442, "loss": 0.0533, "grad_norm": 21.523618698120117, "learning_rate": 7.118181818181819e-06, "num_tokens": 1681936.0, "completions/mean_length": 36.875, "completions/min_length": 32.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.875, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.5032766461372375, "rewards/meter/std": 0.41357243061065674, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9757304191589355, "rewards/repeat_soft/std": 0.02462381310760975, "rewards/judge_quality/mean": 0.8575000166893005, "rewards/judge_quality/std": 0.176776722073555, "rewards/total_composite/mean": 0.6085785627365112, "rewards/total_composite/std": 0.22026771306991577, "reward": 0.6085785627365112, "reward_std": 0.22026769816875458, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16467581689357758, "sampling/sampling_logp_difference/max": 1.5548391342163086, "sampling/importance_sampling_ratio/min": 0.21122334897518158, "sampling/importance_sampling_ratio/mean": 1.0285831689834595, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8324597105383873, "clip_ratio/low_mean": 0.08368434477597475, "clip_ratio/low_min": 0.08368434477597475, "clip_ratio/high_mean": 0.03679234907031059, "clip_ratio/high_max": 0.03679234907031059, "clip_ratio/region_mean": 0.12047669384628534, "reward_total_mean": 0.6085785627365112, "reward_meter_mean": 0.5032766461372375, "reward_meter_std": 0.41357243061065674, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9757304191589355, "reward_repeat_soft_std": 0.02462381310760975, "reward_judge_quality_mean": 0.8575000166893005, "reward_judge_quality_std": 0.176776722073555, "reward_total_composite_mean": 0.6085785627365112, "reward_total_composite_std": 0.22026771306991577} {"timestamp_utc": "2026-04-13T09:45:35Z", "mode": "train", "global_step": 953, "epoch": 0.0957307885484681, "loss": 0.0129, "grad_norm": 23.9711971282959, "learning_rate": 7.115151515151516e-06, "num_tokens": 1683227.0, "completions/mean_length": 22.375, "completions/min_length": 18.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.375, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.6797617673873901, "rewards/meter/std": 0.3961387574672699, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9248495101928711, "rewards/repeat_soft/std": 0.04800232872366905, "rewards/judge_quality/mean": 0.41749998927116394, "rewards/judge_quality/std": 0.06902380287647247, "rewards/total_composite/mean": 0.5333417654037476, "rewards/total_composite/std": 0.11237462610006332, "reward": 0.5333417654037476, "reward_std": 0.11237462610006332, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1608961820602417, "sampling/sampling_logp_difference/max": 1.4233884811401367, "sampling/importance_sampling_ratio/min": 0.24089635908603668, "sampling/importance_sampling_ratio/mean": 1.0058501958847046, "sampling/importance_sampling_ratio/max": 1.7986422777175903, "entropy": 1.142195574939251, "clip_ratio/low_mean": 0.037638889625668526, "clip_ratio/low_min": 0.037638889625668526, "clip_ratio/high_mean": 0.11143544130027294, "clip_ratio/high_max": 0.11143544130027294, "clip_ratio/region_mean": 0.14907433092594147, "reward_total_mean": 0.5333417654037476, "reward_meter_mean": 0.6797617673873901, "reward_meter_std": 0.3961387574672699, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9248495101928711, "reward_repeat_soft_std": 0.04800232872366905, "reward_judge_quality_mean": 0.41749998927116394, "reward_judge_quality_std": 0.06902380287647247, "reward_total_composite_mean": 0.5333417654037476, "reward_total_composite_std": 0.11237462610006332} {"timestamp_utc": "2026-04-13T09:45:46Z", "mode": "train", "global_step": 954, "epoch": 0.0958312405826218, "loss": -0.082, "grad_norm": 3.38273286819458, "learning_rate": 7.1121212121212125e-06, "num_tokens": 1684714.0, "completions/mean_length": 95.875, "completions/min_length": 33.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 36.42857360839844, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.6658740043640137, "rewards/meter/std": 0.37135615944862366, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9352182745933533, "rewards/repeat_soft/std": 0.09705035388469696, "rewards/judge_quality/mean": 0.5400000214576721, "rewards/judge_quality/std": 0.3381884694099426, "rewards/total_composite/mean": 0.569175124168396, "rewards/total_composite/std": 0.29871445894241333, "reward": 0.569175124168396, "reward_std": 0.29871445894241333, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13497833907604218, "sampling/sampling_logp_difference/max": 2.0144739151000977, "sampling/importance_sampling_ratio/min": 0.13339056074619293, "sampling/importance_sampling_ratio/mean": 1.0093296766281128, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7048365473747253, "clip_ratio/low_mean": 0.04988268483430147, "clip_ratio/low_min": 0.04988268483430147, "clip_ratio/high_mean": 0.06866123713552952, "clip_ratio/high_max": 0.06866123713552952, "clip_ratio/region_mean": 0.11854392196983099, "reward_total_mean": 0.569175124168396, "reward_meter_mean": 0.6658740043640137, "reward_meter_std": 0.37135615944862366, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9352182745933533, "reward_repeat_soft_std": 0.09705035388469696, "reward_judge_quality_mean": 0.5400000214576721, "reward_judge_quality_std": 0.3381884694099426, "reward_total_composite_mean": 0.569175124168396, "reward_total_composite_std": 0.29871445894241333} {"timestamp_utc": "2026-04-13T09:45:53Z", "mode": "train", "global_step": 955, "epoch": 0.09593169261677549, "loss": 0.0058, "grad_norm": 11.505900382995605, "learning_rate": 7.10909090909091e-06, "num_tokens": 1686808.0, "completions/mean_length": 79.75, "completions/min_length": 72.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.75, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.1853046715259552, "rewards/meter/std": 0.05850255861878395, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8830885887145996, "rewards/repeat_soft/std": 0.044910505414009094, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.34127578139305115, "rewards/total_composite/std": 0.021256819367408752, "reward": 0.34127578139305115, "reward_std": 0.0212568212300539, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15823736786842346, "sampling/sampling_logp_difference/max": 1.647247314453125, "sampling/importance_sampling_ratio/min": 0.19257928431034088, "sampling/importance_sampling_ratio/mean": 1.0165362358093262, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.031076304614544, "clip_ratio/low_mean": 0.05299842543900013, "clip_ratio/low_min": 0.05299842543900013, "clip_ratio/high_mean": 0.09335256181657314, "clip_ratio/high_max": 0.09335256181657314, "clip_ratio/region_mean": 0.14635098725557327, "reward_total_mean": 0.34127578139305115, "reward_meter_mean": 0.1853046715259552, "reward_meter_std": 0.05850255861878395, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8830885887145996, "reward_repeat_soft_std": 0.044910505414009094, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.34127578139305115, "reward_total_composite_std": 0.021256819367408752} {"timestamp_utc": "2026-04-13T09:46:00Z", "mode": "train", "global_step": 956, "epoch": 0.09603214465092919, "loss": 0.0546, "grad_norm": 8.847649574279785, "learning_rate": 7.106060606060606e-06, "num_tokens": 1688962.0, "completions/mean_length": 93.25, "completions/min_length": 85.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.25, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.9277536273002625, "rewards/meter/std": 0.129310742020607, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9281121492385864, "rewards/repeat_soft/std": 0.056305643171072006, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5524935722351074, "rewards/total_composite/std": 0.03305196017026901, "reward": 0.5524935722351074, "reward_std": 0.03305196762084961, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1509913206100464, "sampling/sampling_logp_difference/max": 1.9407892227172852, "sampling/importance_sampling_ratio/min": 0.14359058439731598, "sampling/importance_sampling_ratio/mean": 1.0155034065246582, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2515875771641731, "clip_ratio/low_mean": 0.019607843831181526, "clip_ratio/low_min": 0.019607843831181526, "clip_ratio/high_mean": 0.1317813778296113, "clip_ratio/high_max": 0.1317813778296113, "clip_ratio/region_mean": 0.15138922166079283, "reward_total_mean": 0.5524935722351074, "reward_meter_mean": 0.9277536273002625, "reward_meter_std": 0.129310742020607, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9281121492385864, "reward_repeat_soft_std": 0.056305643171072006, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5524935722351074, "reward_total_composite_std": 0.03305196017026901} {"timestamp_utc": "2026-04-13T09:46:05Z", "mode": "train", "global_step": 957, "epoch": 0.09613259668508287, "loss": -0.0271, "grad_norm": 19.07744026184082, "learning_rate": 7.103030303030304e-06, "num_tokens": 1690193.0, "completions/mean_length": 24.875, "completions/min_length": 20.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.875, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.456256628036499, "rewards/meter/std": 0.4533495604991913, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9200776815414429, "rewards/repeat_soft/std": 0.051106806844472885, "rewards/judge_quality/mean": 0.4424999952316284, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.4714059829711914, "rewards/total_composite/std": 0.13499127328395844, "reward": 0.4714059829711914, "reward_std": 0.13499125838279724, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16945433616638184, "sampling/sampling_logp_difference/max": 0.8939728736877441, "sampling/importance_sampling_ratio/min": 0.4090275168418884, "sampling/importance_sampling_ratio/mean": 1.0692710876464844, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5706066191196442, "clip_ratio/low_mean": 0.10254121199250221, "clip_ratio/low_min": 0.10254121199250221, "clip_ratio/high_mean": 0.06304713990539312, "clip_ratio/high_max": 0.06304713990539312, "clip_ratio/region_mean": 0.16558835189789534, "reward_total_mean": 0.4714059829711914, "reward_meter_mean": 0.456256628036499, "reward_meter_std": 0.4533495604991913, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9200776815414429, "reward_repeat_soft_std": 0.051106806844472885, "reward_judge_quality_mean": 0.4424999952316284, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.4714059829711914, "reward_total_composite_std": 0.13499127328395844} {"timestamp_utc": "2026-04-13T09:46:11Z", "mode": "train", "global_step": 958, "epoch": 0.09623304871923656, "loss": 0.0107, "grad_norm": 13.06394100189209, "learning_rate": 7.100000000000001e-06, "num_tokens": 1691815.0, "completions/mean_length": 42.75, "completions/min_length": 40.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.75, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.6433264017105103, "rewards/meter/std": 0.31973007321357727, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9257329106330872, "rewards/repeat_soft/std": 0.08968224376440048, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.11667262762784958, "rewards/total_composite/mean": 0.53730309009552, "rewards/total_composite/std": 0.1151580959558487, "reward": 0.53730309009552, "reward_std": 0.1151580885052681, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14228375256061554, "sampling/sampling_logp_difference/max": 2.067209005355835, "sampling/importance_sampling_ratio/min": 0.1265384554862976, "sampling/importance_sampling_ratio/mean": 1.0139461755752563, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.035200595855713, "clip_ratio/low_mean": 0.08142944891005754, "clip_ratio/low_min": 0.08142944891005754, "clip_ratio/high_mean": 0.07406162843108177, "clip_ratio/high_max": 0.07406162843108177, "clip_ratio/region_mean": 0.15549107734113932, "reward_total_mean": 0.53730309009552, "reward_meter_mean": 0.6433264017105103, "reward_meter_std": 0.31973007321357727, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9257329106330872, "reward_repeat_soft_std": 0.08968224376440048, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.11667262762784958, "reward_total_composite_mean": 0.53730309009552, "reward_total_composite_std": 0.1151580959558487} {"timestamp_utc": "2026-04-13T09:46:23Z", "mode": "train", "global_step": 959, "epoch": 0.09633350075339026, "loss": -0.056, "grad_norm": 1.32066810131073, "learning_rate": 7.096969696969698e-06, "num_tokens": 1693141.0, "completions/mean_length": 144.75, "completions/min_length": 18.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 22.33333396911621, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 25.0, "rewards/meter/mean": 0.5230863690376282, "rewards/meter/std": 0.36974969506263733, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9671874642372131, "rewards/repeat_soft/std": 0.013258260674774647, "rewards/judge_quality/mean": 0.4087499976158142, "rewards/judge_quality/std": 0.27445724606513977, "rewards/total_composite/mean": 0.43389588594436646, "rewards/total_composite/std": 0.27731138467788696, "reward": 0.43389588594436646, "reward_std": 0.27731141448020935, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1253676563501358, "sampling/sampling_logp_difference/max": 1.0962228775024414, "sampling/importance_sampling_ratio/min": 0.33413076400756836, "sampling/importance_sampling_ratio/mean": 1.0182522535324097, "sampling/importance_sampling_ratio/max": 1.468483567237854, "entropy": 0.8347451686859131, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09764190902933478, "clip_ratio/high_max": 0.09764190902933478, "clip_ratio/region_mean": 0.09764190902933478, "reward_total_mean": 0.43389588594436646, "reward_meter_mean": 0.5230863690376282, "reward_meter_std": 0.36974969506263733, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9671874642372131, "reward_repeat_soft_std": 0.013258260674774647, "reward_judge_quality_mean": 0.4087499976158142, "reward_judge_quality_std": 0.27445724606513977, "reward_total_composite_mean": 0.43389588594436646, "reward_total_composite_std": 0.27731138467788696} {"timestamp_utc": "2026-04-13T09:46:29Z", "mode": "train", "global_step": 960, "epoch": 0.09643395278754395, "loss": 0.0433, "grad_norm": 9.702564239501953, "learning_rate": 7.093939393939394e-06, "num_tokens": 1694803.0, "completions/mean_length": 42.75, "completions/min_length": 38.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.75, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.8904345035552979, "rewards/meter/std": 0.18871240317821503, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9875833988189697, "rewards/repeat_soft/std": 0.006782568525522947, "rewards/judge_quality/mean": 0.5349999666213989, "rewards/judge_quality/std": 0.18431341648101807, "rewards/total_composite/mean": 0.6410684585571289, "rewards/total_composite/std": 0.06271853297948837, "reward": 0.6410684585571289, "reward_std": 0.06271851807832718, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16253142058849335, "sampling/sampling_logp_difference/max": 1.612654209136963, "sampling/importance_sampling_ratio/min": 0.1993577778339386, "sampling/importance_sampling_ratio/mean": 1.015665888786316, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9276193827390671, "clip_ratio/low_mean": 0.12437493074685335, "clip_ratio/low_min": 0.12437493074685335, "clip_ratio/high_mean": 0.027439024299383163, "clip_ratio/high_max": 0.027439024299383163, "clip_ratio/region_mean": 0.15181395504623652, "reward_total_mean": 0.6410684585571289, "reward_meter_mean": 0.8904345035552979, "reward_meter_std": 0.18871240317821503, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9875833988189697, "reward_repeat_soft_std": 0.006782568525522947, "reward_judge_quality_mean": 0.5349999666213989, "reward_judge_quality_std": 0.18431341648101807, "reward_total_composite_mean": 0.6410684585571289, "reward_total_composite_std": 0.06271853297948837} {"timestamp_utc": "2026-04-13T09:46:35Z", "mode": "train", "global_step": 961, "epoch": 0.09653440482169764, "loss": 0.0405, "grad_norm": 13.252286911010742, "learning_rate": 7.0909090909090916e-06, "num_tokens": 1696555.0, "completions/mean_length": 37.0, "completions/min_length": 32.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.47514355182647705, "rewards/meter/std": 0.3408444821834564, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9729394316673279, "rewards/repeat_soft/std": 0.029924456030130386, "rewards/judge_quality/mean": 0.8575000166893005, "rewards/judge_quality/std": 0.176776722073555, "rewards/total_composite/mean": 0.614441990852356, "rewards/total_composite/std": 0.21273763477802277, "reward": 0.614441990852356, "reward_std": 0.21273763477802277, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1581917554140091, "sampling/sampling_logp_difference/max": 1.3821625709533691, "sampling/importance_sampling_ratio/min": 0.27115383744239807, "sampling/importance_sampling_ratio/mean": 1.0278681516647339, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9341058731079102, "clip_ratio/low_mean": 0.07212148141115904, "clip_ratio/low_min": 0.07212148141115904, "clip_ratio/high_mean": 0.07070096489042044, "clip_ratio/high_max": 0.07070096489042044, "clip_ratio/region_mean": 0.14282244630157948, "reward_total_mean": 0.614441990852356, "reward_meter_mean": 0.47514355182647705, "reward_meter_std": 0.3408444821834564, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9729394316673279, "reward_repeat_soft_std": 0.029924456030130386, "reward_judge_quality_mean": 0.8575000166893005, "reward_judge_quality_std": 0.176776722073555, "reward_total_composite_mean": 0.614441990852356, "reward_total_composite_std": 0.21273763477802277} {"timestamp_utc": "2026-04-13T09:46:41Z", "mode": "train", "global_step": 962, "epoch": 0.09663485685585133, "loss": -0.0148, "grad_norm": 14.771444320678711, "learning_rate": 7.087878787878788e-06, "num_tokens": 1698103.0, "completions/mean_length": 40.5, "completions/min_length": 34.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.5, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.7526412010192871, "rewards/meter/std": 0.4042608439922333, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9692273139953613, "rewards/repeat_soft/std": 0.039702706038951874, "rewards/judge_quality/mean": 0.6737500429153442, "rewards/judge_quality/std": 0.263435423374176, "rewards/total_composite/mean": 0.6720747947692871, "rewards/total_composite/std": 0.22597767412662506, "reward": 0.6720747947692871, "reward_std": 0.22597767412662506, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14504121243953705, "sampling/sampling_logp_difference/max": 1.164337158203125, "sampling/importance_sampling_ratio/min": 0.31212949752807617, "sampling/importance_sampling_ratio/mean": 1.0402779579162598, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1788065060973167, "clip_ratio/low_mean": 0.08813421241939068, "clip_ratio/low_min": 0.08813421241939068, "clip_ratio/high_mean": 0.06852390244603157, "clip_ratio/high_max": 0.06852390244603157, "clip_ratio/region_mean": 0.15665811486542225, "reward_total_mean": 0.6720747947692871, "reward_meter_mean": 0.7526412010192871, "reward_meter_std": 0.4042608439922333, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9692273139953613, "reward_repeat_soft_std": 0.039702706038951874, "reward_judge_quality_mean": 0.6737500429153442, "reward_judge_quality_std": 0.263435423374176, "reward_total_composite_mean": 0.6720747947692871, "reward_total_composite_std": 0.22597767412662506} {"timestamp_utc": "2026-04-13T09:46:48Z", "mode": "train", "global_step": 963, "epoch": 0.09673530889000502, "loss": -0.0079, "grad_norm": 12.6827974319458, "learning_rate": 7.084848484848485e-06, "num_tokens": 1699921.0, "completions/mean_length": 50.25, "completions/min_length": 44.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.25, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.6226515173912048, "rewards/meter/std": 0.3655487895011902, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9648508429527283, "rewards/repeat_soft/std": 0.0903712809085846, "rewards/judge_quality/mean": 0.5275000333786011, "rewards/judge_quality/std": 0.18873640894889832, "rewards/total_composite/mean": 0.5672609806060791, "rewards/total_composite/std": 0.1732962727546692, "reward": 0.5672609806060791, "reward_std": 0.1732962727546692, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1483352780342102, "sampling/sampling_logp_difference/max": 1.4136152267456055, "sampling/importance_sampling_ratio/min": 0.24326223134994507, "sampling/importance_sampling_ratio/mean": 1.0176331996917725, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3203382641077042, "clip_ratio/low_mean": 0.10926157608628273, "clip_ratio/low_min": 0.10926157608628273, "clip_ratio/high_mean": 0.04935689456760883, "clip_ratio/high_max": 0.04935689456760883, "clip_ratio/region_mean": 0.15861847065389156, "reward_total_mean": 0.5672609806060791, "reward_meter_mean": 0.6226515173912048, "reward_meter_std": 0.3655487895011902, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9648508429527283, "reward_repeat_soft_std": 0.0903712809085846, "reward_judge_quality_mean": 0.5275000333786011, "reward_judge_quality_std": 0.18873640894889832, "reward_total_composite_mean": 0.5672609806060791, "reward_total_composite_std": 0.1732962727546692} {"timestamp_utc": "2026-04-13T09:46:54Z", "mode": "train", "global_step": 964, "epoch": 0.09683576092415871, "loss": 0.0154, "grad_norm": 13.873187065124512, "learning_rate": 7.081818181818182e-06, "num_tokens": 1701561.0, "completions/mean_length": 41.0, "completions/min_length": 37.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.0, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.9538505673408508, "rewards/meter/std": 0.05888332799077034, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9307888150215149, "rewards/repeat_soft/std": 0.06291955709457397, "rewards/judge_quality/mean": 0.46875, "rewards/judge_quality/std": 0.10260014235973358, "rewards/total_composite/mean": 0.6308960914611816, "rewards/total_composite/std": 0.07448755949735641, "reward": 0.6308960914611816, "reward_std": 0.07448754459619522, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12618441879749298, "sampling/sampling_logp_difference/max": 1.232431411743164, "sampling/importance_sampling_ratio/min": 0.29158276319503784, "sampling/importance_sampling_ratio/mean": 1.0101318359375, "sampling/importance_sampling_ratio/max": 1.8735744953155518, "entropy": 1.036969743669033, "clip_ratio/low_mean": 0.054261722369119525, "clip_ratio/low_min": 0.054261722369119525, "clip_ratio/high_mean": 0.025240384973585606, "clip_ratio/high_max": 0.025240384973585606, "clip_ratio/region_mean": 0.07950210734270513, "reward_total_mean": 0.6308960914611816, "reward_meter_mean": 0.9538505673408508, "reward_meter_std": 0.05888332799077034, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9307888150215149, "reward_repeat_soft_std": 0.06291955709457397, "reward_judge_quality_mean": 0.46875, "reward_judge_quality_std": 0.10260014235973358, "reward_total_composite_mean": 0.6308960914611816, "reward_total_composite_std": 0.07448755949735641} {"timestamp_utc": "2026-04-13T09:47:05Z", "mode": "train", "global_step": 965, "epoch": 0.09693621295831241, "loss": -0.096, "grad_norm": 2.3022987842559814, "learning_rate": 7.07878787878788e-06, "num_tokens": 1703041.0, "completions/mean_length": 160.0, "completions/min_length": 35.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 42.66666793823242, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.4400330185890198, "rewards/meter/std": 0.31648334860801697, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9835957288742065, "rewards/repeat_soft/std": 0.015569952316582203, "rewards/judge_quality/mean": 0.3762499988079071, "rewards/judge_quality/std": 0.2775370180606842, "rewards/total_composite/mean": 0.36539924144744873, "rewards/total_composite/std": 0.23746727406978607, "reward": 0.36539924144744873, "reward_std": 0.23746725916862488, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14925897121429443, "sampling/sampling_logp_difference/max": 1.7131309509277344, "sampling/importance_sampling_ratio/min": 0.18030039966106415, "sampling/importance_sampling_ratio/mean": 1.0325299501419067, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5369725301861763, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09194457437843084, "clip_ratio/high_max": 0.09194457437843084, "clip_ratio/region_mean": 0.09194457437843084, "reward_total_mean": 0.36539924144744873, "reward_meter_mean": 0.4400330185890198, "reward_meter_std": 0.31648334860801697, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9835957288742065, "reward_repeat_soft_std": 0.015569952316582203, "reward_judge_quality_mean": 0.3762499988079071, "reward_judge_quality_std": 0.2775370180606842, "reward_total_composite_mean": 0.36539924144744873, "reward_total_composite_std": 0.23746727406978607} {"timestamp_utc": "2026-04-13T09:47:12Z", "mode": "train", "global_step": 966, "epoch": 0.0970366649924661, "loss": 0.0334, "grad_norm": 13.992204666137695, "learning_rate": 7.075757575757576e-06, "num_tokens": 1704582.0, "completions/mean_length": 35.625, "completions/min_length": 31.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.625, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.745686411857605, "rewards/meter/std": 0.34231388568878174, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9419390559196472, "rewards/repeat_soft/std": 0.057287752628326416, "rewards/judge_quality/mean": 0.39750000834465027, "rewards/judge_quality/std": 0.10110107809305191, "rewards/total_composite/mean": 0.5078482031822205, "rewards/total_composite/std": 0.21228386461734772, "reward": 0.5078482031822205, "reward_std": 0.21228386461734772, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14791765809059143, "sampling/sampling_logp_difference/max": 1.932054042816162, "sampling/importance_sampling_ratio/min": 0.14485037326812744, "sampling/importance_sampling_ratio/mean": 0.9893924593925476, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.854806199669838, "clip_ratio/low_mean": 0.020003555342555046, "clip_ratio/low_min": 0.020003555342555046, "clip_ratio/high_mean": 0.11828875355422497, "clip_ratio/high_max": 0.11828875355422497, "clip_ratio/region_mean": 0.13829230889678001, "reward_total_mean": 0.5078482031822205, "reward_meter_mean": 0.745686411857605, "reward_meter_std": 0.34231388568878174, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9419390559196472, "reward_repeat_soft_std": 0.057287752628326416, "reward_judge_quality_mean": 0.39750000834465027, "reward_judge_quality_std": 0.10110107809305191, "reward_total_composite_mean": 0.5078482031822205, "reward_total_composite_std": 0.21228386461734772} {"timestamp_utc": "2026-04-13T09:47:18Z", "mode": "train", "global_step": 967, "epoch": 0.09713711702661978, "loss": 0.0013, "grad_norm": 9.609872817993164, "learning_rate": 7.072727272727273e-06, "num_tokens": 1706536.0, "completions/mean_length": 70.25, "completions/min_length": 64.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.25, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.5268456935882568, "rewards/meter/std": 0.2801974415779114, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9801520109176636, "rewards/repeat_soft/std": 0.029769711196422577, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.13695022463798523, "rewards/total_composite/mean": 0.4549916684627533, "rewards/total_composite/std": 0.22096389532089233, "reward": 0.4549916684627533, "reward_std": 0.22096388041973114, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13744767010211945, "sampling/sampling_logp_difference/max": 1.8090505599975586, "sampling/importance_sampling_ratio/min": 0.16380958259105682, "sampling/importance_sampling_ratio/mean": 1.023453950881958, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6548539251089096, "clip_ratio/low_mean": 0.041847581043839455, "clip_ratio/low_min": 0.041847581043839455, "clip_ratio/high_mean": 0.06069137854501605, "clip_ratio/high_max": 0.06069137854501605, "clip_ratio/region_mean": 0.1025389595888555, "reward_total_mean": 0.4549916684627533, "reward_meter_mean": 0.5268456935882568, "reward_meter_std": 0.2801974415779114, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9801520109176636, "reward_repeat_soft_std": 0.029769711196422577, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.13695022463798523, "reward_total_composite_mean": 0.4549916684627533, "reward_total_composite_std": 0.22096389532089233} {"timestamp_utc": "2026-04-13T09:47:25Z", "mode": "train", "global_step": 968, "epoch": 0.09723756906077348, "loss": -0.0144, "grad_norm": 13.045398712158203, "learning_rate": 7.06969696969697e-06, "num_tokens": 1708262.0, "completions/mean_length": 50.75, "completions/min_length": 40.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.75, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.5692801475524902, "rewards/meter/std": 0.3768622875213623, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9812266826629639, "rewards/repeat_soft/std": 0.024538792669773102, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.5042853951454163, "rewards/total_composite/std": 0.10177413374185562, "reward": 0.5042853951454163, "reward_std": 0.10177412629127502, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15098918974399567, "sampling/sampling_logp_difference/max": 1.3089690208435059, "sampling/importance_sampling_ratio/min": 0.27009838819503784, "sampling/importance_sampling_ratio/mean": 1.0211642980575562, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2397021055221558, "clip_ratio/low_mean": 0.05614245776087046, "clip_ratio/low_min": 0.05614245776087046, "clip_ratio/high_mean": 0.06984651274979115, "clip_ratio/high_max": 0.06984651274979115, "clip_ratio/region_mean": 0.1259889705106616, "reward_total_mean": 0.5042853951454163, "reward_meter_mean": 0.5692801475524902, "reward_meter_std": 0.3768622875213623, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9812266826629639, "reward_repeat_soft_std": 0.024538792669773102, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.5042853951454163, "reward_total_composite_std": 0.10177413374185562} {"timestamp_utc": "2026-04-13T09:47:31Z", "mode": "train", "global_step": 969, "epoch": 0.09733802109492717, "loss": -0.0054, "grad_norm": 13.439970970153809, "learning_rate": 7.066666666666667e-06, "num_tokens": 1709864.0, "completions/mean_length": 50.25, "completions/min_length": 47.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.25, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.8206380009651184, "rewards/meter/std": 0.33159759640693665, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9912235736846924, "rewards/repeat_soft/std": 0.008905306458473206, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.534211277961731, "rewards/total_composite/std": 0.21942123770713806, "reward": 0.534211277961731, "reward_std": 0.21942123770713806, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1369597315788269, "sampling/sampling_logp_difference/max": 1.4509609937667847, "sampling/importance_sampling_ratio/min": 0.2343449741601944, "sampling/importance_sampling_ratio/mean": 1.0040875673294067, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9222754910588264, "clip_ratio/low_mean": 0.02849544119089842, "clip_ratio/low_min": 0.02849544119089842, "clip_ratio/high_mean": 0.10500968247652054, "clip_ratio/high_max": 0.10500968247652054, "clip_ratio/region_mean": 0.13350512366741896, "reward_total_mean": 0.534211277961731, "reward_meter_mean": 0.8206380009651184, "reward_meter_std": 0.33159759640693665, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9912235736846924, "reward_repeat_soft_std": 0.008905306458473206, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.534211277961731, "reward_total_composite_std": 0.21942123770713806} {"timestamp_utc": "2026-04-13T09:47:42Z", "mode": "train", "global_step": 970, "epoch": 0.09743847312908087, "loss": -0.1169, "grad_norm": 2.641662836074829, "learning_rate": 7.063636363636365e-06, "num_tokens": 1711459.0, "completions/mean_length": 102.375, "completions/min_length": 39.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 43.85714340209961, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.6911942958831787, "rewards/meter/std": 0.3835291266441345, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9280762672424316, "rewards/repeat_soft/std": 0.045472707599401474, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.13274572789669037, "rewards/total_composite/mean": 0.4747639000415802, "rewards/total_composite/std": 0.21479584276676178, "reward": 0.4747639000415802, "reward_std": 0.21479582786560059, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13141244649887085, "sampling/sampling_logp_difference/max": 1.3056834936141968, "sampling/importance_sampling_ratio/min": 0.27098727226257324, "sampling/importance_sampling_ratio/mean": 1.0224891901016235, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7379521951079369, "clip_ratio/low_mean": 0.033444148022681475, "clip_ratio/low_min": 0.033444148022681475, "clip_ratio/high_mean": 0.09196178521960974, "clip_ratio/high_max": 0.09196178521960974, "clip_ratio/region_mean": 0.1254059332422912, "reward_total_mean": 0.4747639000415802, "reward_meter_mean": 0.6911942958831787, "reward_meter_std": 0.3835291266441345, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9280762672424316, "reward_repeat_soft_std": 0.045472707599401474, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.13274572789669037, "reward_total_composite_mean": 0.4747639000415802, "reward_total_composite_std": 0.21479584276676178} {"timestamp_utc": "2026-04-13T09:47:48Z", "mode": "train", "global_step": 971, "epoch": 0.09753892516323455, "loss": 0.0848, "grad_norm": 14.68692684173584, "learning_rate": 7.060606060606061e-06, "num_tokens": 1712834.0, "completions/mean_length": 25.875, "completions/min_length": 22.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.875, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.14169715344905853, "rewards/meter/std": 0.2590996026992798, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9521924257278442, "rewards/repeat_soft/std": 0.018484456464648247, "rewards/judge_quality/mean": 0.6225000023841858, "rewards/judge_quality/std": 0.24656209349632263, "rewards/total_composite/mean": 0.3912311792373657, "rewards/total_composite/std": 0.07280620187520981, "reward": 0.3912311792373657, "reward_std": 0.07280620187520981, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1792679876089096, "sampling/sampling_logp_difference/max": 2.050839424133301, "sampling/importance_sampling_ratio/min": 0.12862688302993774, "sampling/importance_sampling_ratio/mean": 1.0485748052597046, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4391480088233948, "clip_ratio/low_mean": 0.10998993832617998, "clip_ratio/low_min": 0.10998993832617998, "clip_ratio/high_mean": 0.04471343941986561, "clip_ratio/high_max": 0.04471343941986561, "clip_ratio/region_mean": 0.1547033777460456, "reward_total_mean": 0.3912311792373657, "reward_meter_mean": 0.14169715344905853, "reward_meter_std": 0.2590996026992798, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9521924257278442, "reward_repeat_soft_std": 0.018484456464648247, "reward_judge_quality_mean": 0.6225000023841858, "reward_judge_quality_std": 0.24656209349632263, "reward_total_composite_mean": 0.3912311792373657, "reward_total_composite_std": 0.07280620187520981} {"timestamp_utc": "2026-04-13T09:47:55Z", "mode": "train", "global_step": 972, "epoch": 0.09763937719738824, "loss": 0.065, "grad_norm": 13.530476570129395, "learning_rate": 7.057575757575759e-06, "num_tokens": 1714822.0, "completions/mean_length": 75.5, "completions/min_length": 67.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.5, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.43695682287216187, "rewards/meter/std": 0.3026445209980011, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9667505025863647, "rewards/repeat_soft/std": 0.025144891813397408, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.4592045247554779, "rewards/total_composite/std": 0.0866062343120575, "reward": 0.4592045247554779, "reward_std": 0.08660624176263809, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16183550655841827, "sampling/sampling_logp_difference/max": 1.2516584396362305, "sampling/importance_sampling_ratio/min": 0.2860300540924072, "sampling/importance_sampling_ratio/mean": 1.0253628492355347, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4893262386322021, "clip_ratio/low_mean": 0.10574120655655861, "clip_ratio/low_min": 0.10574120655655861, "clip_ratio/high_mean": 0.06007573939859867, "clip_ratio/high_max": 0.06007573939859867, "clip_ratio/region_mean": 0.16581694595515728, "reward_total_mean": 0.4592045247554779, "reward_meter_mean": 0.43695682287216187, "reward_meter_std": 0.3026445209980011, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9667505025863647, "reward_repeat_soft_std": 0.025144891813397408, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.4592045247554779, "reward_total_composite_std": 0.0866062343120575} {"timestamp_utc": "2026-04-13T09:48:01Z", "mode": "train", "global_step": 973, "epoch": 0.09773982923154194, "loss": 0.091, "grad_norm": 11.856672286987305, "learning_rate": 7.054545454545455e-06, "num_tokens": 1716496.0, "completions/mean_length": 44.25, "completions/min_length": 31.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.25, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.9789201021194458, "rewards/meter/std": 0.013528779149055481, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.938413679599762, "rewards/repeat_soft/std": 0.04366263374686241, "rewards/judge_quality/mean": 0.5649999976158142, "rewards/judge_quality/std": 0.194054514169693, "rewards/total_composite/mean": 0.7009687423706055, "rewards/total_composite/std": 0.13211536407470703, "reward": 0.7009687423706055, "reward_std": 0.13211536407470703, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14514295756816864, "sampling/sampling_logp_difference/max": 1.3580703735351562, "sampling/importance_sampling_ratio/min": 0.25715652108192444, "sampling/importance_sampling_ratio/mean": 1.0207626819610596, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2455740720033646, "clip_ratio/low_mean": 0.06271467031911016, "clip_ratio/low_min": 0.06271467031911016, "clip_ratio/high_mean": 0.06903463043272495, "clip_ratio/high_max": 0.06903463043272495, "clip_ratio/region_mean": 0.1317493007518351, "reward_total_mean": 0.7009687423706055, "reward_meter_mean": 0.9789201021194458, "reward_meter_std": 0.013528779149055481, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.938413679599762, "reward_repeat_soft_std": 0.04366263374686241, "reward_judge_quality_mean": 0.5649999976158142, "reward_judge_quality_std": 0.194054514169693, "reward_total_composite_mean": 0.7009687423706055, "reward_total_composite_std": 0.13211536407470703} {"timestamp_utc": "2026-04-13T09:48:07Z", "mode": "train", "global_step": 974, "epoch": 0.09784028126569563, "loss": 0.0094, "grad_norm": 12.079497337341309, "learning_rate": 7.0515151515151525e-06, "num_tokens": 1718125.0, "completions/mean_length": 51.625, "completions/min_length": 40.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.625, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.920598030090332, "rewards/meter/std": 0.11000407487154007, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9650930762290955, "rewards/repeat_soft/std": 0.047561854124069214, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.22720351815223694, "rewards/total_composite/mean": 0.664365291595459, "rewards/total_composite/std": 0.0953650251030922, "reward": 0.664365291595459, "reward_std": 0.09536503255367279, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13764303922653198, "sampling/sampling_logp_difference/max": 1.8628220558166504, "sampling/importance_sampling_ratio/min": 0.1552339345216751, "sampling/importance_sampling_ratio/mean": 1.0276992321014404, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1694976165890694, "clip_ratio/low_mean": 0.0747088547796011, "clip_ratio/low_min": 0.0747088547796011, "clip_ratio/high_mean": 0.03707483038306236, "clip_ratio/high_max": 0.03707483038306236, "clip_ratio/region_mean": 0.11178368516266346, "reward_total_mean": 0.664365291595459, "reward_meter_mean": 0.920598030090332, "reward_meter_std": 0.11000407487154007, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9650930762290955, "reward_repeat_soft_std": 0.047561854124069214, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.22720351815223694, "reward_total_composite_mean": 0.664365291595459, "reward_total_composite_std": 0.0953650251030922} {"timestamp_utc": "2026-04-13T09:48:18Z", "mode": "train", "global_step": 975, "epoch": 0.09794073329984933, "loss": -0.1509, "grad_norm": 2.8824329376220703, "learning_rate": 7.048484848484849e-06, "num_tokens": 1719952.0, "completions/mean_length": 124.375, "completions/min_length": 59.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 69.0, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.709470272064209, "rewards/meter/std": 0.3198787271976471, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9301737546920776, "rewards/repeat_soft/std": 0.04480618238449097, "rewards/judge_quality/mean": 0.44874998927116394, "rewards/judge_quality/std": 0.2105392962694168, "rewards/total_composite/mean": 0.5060956478118896, "rewards/total_composite/std": 0.23397000133991241, "reward": 0.5060956478118896, "reward_std": 0.23396998643875122, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14583653211593628, "sampling/sampling_logp_difference/max": 1.4082255363464355, "sampling/importance_sampling_ratio/min": 0.24457688629627228, "sampling/importance_sampling_ratio/mean": 1.031238317489624, "sampling/importance_sampling_ratio/max": 1.9890636205673218, "entropy": 1.0118075907230377, "clip_ratio/low_mean": 0.021739130839705467, "clip_ratio/low_min": 0.021739130839705467, "clip_ratio/high_mean": 0.10367511957883835, "clip_ratio/high_max": 0.10367511957883835, "clip_ratio/region_mean": 0.12541425041854382, "reward_total_mean": 0.5060956478118896, "reward_meter_mean": 0.709470272064209, "reward_meter_std": 0.3198787271976471, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9301737546920776, "reward_repeat_soft_std": 0.04480618238449097, "reward_judge_quality_mean": 0.44874998927116394, "reward_judge_quality_std": 0.2105392962694168, "reward_total_composite_mean": 0.5060956478118896, "reward_total_composite_std": 0.23397000133991241} {"timestamp_utc": "2026-04-13T09:48:24Z", "mode": "train", "global_step": 976, "epoch": 0.09804118533400301, "loss": 0.0768, "grad_norm": 12.867921829223633, "learning_rate": 7.045454545454546e-06, "num_tokens": 1721600.0, "completions/mean_length": 52.0, "completions/min_length": 46.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.0, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.7738261222839355, "rewards/meter/std": 0.3358989655971527, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9791769981384277, "rewards/repeat_soft/std": 0.017892280593514442, "rewards/judge_quality/mean": 0.5487500429153442, "rewards/judge_quality/std": 0.22937415540218353, "rewards/total_composite/mean": 0.5318248271942139, "rewards/total_composite/std": 0.26320815086364746, "reward": 0.5318248271942139, "reward_std": 0.26320815086364746, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.160322904586792, "sampling/sampling_logp_difference/max": 1.9544696807861328, "sampling/importance_sampling_ratio/min": 0.1416395604610443, "sampling/importance_sampling_ratio/mean": 1.0299495458602905, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2601183354854584, "clip_ratio/low_mean": 0.046510683838278055, "clip_ratio/low_min": 0.046510683838278055, "clip_ratio/high_mean": 0.078507199883461, "clip_ratio/high_max": 0.078507199883461, "clip_ratio/region_mean": 0.12501788372173905, "reward_total_mean": 0.5318248271942139, "reward_meter_mean": 0.7738261222839355, "reward_meter_std": 0.3358989655971527, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9791769981384277, "reward_repeat_soft_std": 0.017892280593514442, "reward_judge_quality_mean": 0.5487500429153442, "reward_judge_quality_std": 0.22937415540218353, "reward_total_composite_mean": 0.5318248271942139, "reward_total_composite_std": 0.26320815086364746} {"timestamp_utc": "2026-04-13T09:48:30Z", "mode": "train", "global_step": 977, "epoch": 0.0981416373681567, "loss": 0.0766, "grad_norm": 16.257875442504883, "learning_rate": 7.0424242424242426e-06, "num_tokens": 1723047.0, "completions/mean_length": 30.875, "completions/min_length": 25.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.875, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9210174083709717, "rewards/meter/std": 0.06121338531374931, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9290193319320679, "rewards/repeat_soft/std": 0.05646272376179695, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.7440645694732666, "rewards/total_composite/std": 0.17584004998207092, "reward": 0.7440645694732666, "reward_std": 0.17584003508090973, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12017352879047394, "sampling/sampling_logp_difference/max": 1.0964205265045166, "sampling/importance_sampling_ratio/min": 0.44648265838623047, "sampling/importance_sampling_ratio/mean": 1.02653968334198, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7153010219335556, "clip_ratio/low_mean": 0.07673509418964386, "clip_ratio/low_min": 0.07673509418964386, "clip_ratio/high_mean": 0.05919354781508446, "clip_ratio/high_max": 0.05919354781508446, "clip_ratio/region_mean": 0.13592864200472832, "reward_total_mean": 0.7440645694732666, "reward_meter_mean": 0.9210174083709717, "reward_meter_std": 0.06121338531374931, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9290193319320679, "reward_repeat_soft_std": 0.05646272376179695, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.7440645694732666, "reward_total_composite_std": 0.17584004998207092} {"timestamp_utc": "2026-04-13T09:48:36Z", "mode": "train", "global_step": 978, "epoch": 0.0982420894023104, "loss": 0.1118, "grad_norm": 12.845317840576172, "learning_rate": 7.039393939393941e-06, "num_tokens": 1724862.0, "completions/mean_length": 51.875, "completions/min_length": 45.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.875, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.5170915126800537, "rewards/meter/std": 0.4167918562889099, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9292737245559692, "rewards/repeat_soft/std": 0.06074531748890877, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.5086134672164917, "rewards/total_composite/std": 0.20576469600200653, "reward": 0.5086134672164917, "reward_std": 0.20576469600200653, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16256125271320343, "sampling/sampling_logp_difference/max": 2.1671018600463867, "sampling/importance_sampling_ratio/min": 0.11450900137424469, "sampling/importance_sampling_ratio/mean": 1.0142372846603394, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0791840329766273, "clip_ratio/low_mean": 0.06558190565556288, "clip_ratio/low_min": 0.06558190565556288, "clip_ratio/high_mean": 0.07605824619531631, "clip_ratio/high_max": 0.07605824619531631, "clip_ratio/region_mean": 0.1416401518508792, "reward_total_mean": 0.5086134672164917, "reward_meter_mean": 0.5170915126800537, "reward_meter_std": 0.4167918562889099, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9292737245559692, "reward_repeat_soft_std": 0.06074531748890877, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.5086134672164917, "reward_total_composite_std": 0.20576469600200653} {"timestamp_utc": "2026-04-13T09:48:42Z", "mode": "train", "global_step": 979, "epoch": 0.09834254143646409, "loss": 0.2517, "grad_norm": 17.47881317138672, "learning_rate": 7.036363636363637e-06, "num_tokens": 1726339.0, "completions/mean_length": 25.625, "completions/min_length": 20.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.625, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.9364637136459351, "rewards/meter/std": 0.11381543427705765, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9130376577377319, "rewards/repeat_soft/std": 0.05440940707921982, "rewards/judge_quality/mean": 0.5012500286102295, "rewards/judge_quality/std": 0.16974246501922607, "rewards/total_composite/mean": 0.6187773942947388, "rewards/total_composite/std": 0.15915215015411377, "reward": 0.6187773942947388, "reward_std": 0.15915215015411377, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13362014293670654, "sampling/sampling_logp_difference/max": 1.0516290664672852, "sampling/importance_sampling_ratio/min": 0.3493681252002716, "sampling/importance_sampling_ratio/mean": 1.01193106174469, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9321458712220192, "clip_ratio/low_mean": 0.06847826205193996, "clip_ratio/low_min": 0.06847826205193996, "clip_ratio/high_mean": 0.10737812984734774, "clip_ratio/high_max": 0.10737812984734774, "clip_ratio/region_mean": 0.1758563918992877, "reward_total_mean": 0.6187773942947388, "reward_meter_mean": 0.9364637136459351, "reward_meter_std": 0.11381543427705765, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9130376577377319, "reward_repeat_soft_std": 0.05440940707921982, "reward_judge_quality_mean": 0.5012500286102295, "reward_judge_quality_std": 0.16974246501922607, "reward_total_composite_mean": 0.6187773942947388, "reward_total_composite_std": 0.15915215015411377} {"timestamp_utc": "2026-04-13T09:48:49Z", "mode": "train", "global_step": 980, "epoch": 0.09844299347061777, "loss": 0.0489, "grad_norm": 9.139081954956055, "learning_rate": 7.033333333333334e-06, "num_tokens": 1728496.0, "completions/mean_length": 90.625, "completions/min_length": 75.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.625, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.7046166062355042, "rewards/meter/std": 0.33246853947639465, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8811008334159851, "rewards/repeat_soft/std": 0.04722922295331955, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.5273675918579102, "rewards/total_composite/std": 0.08341158181428909, "reward": 0.5273675918579102, "reward_std": 0.0834115743637085, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13319680094718933, "sampling/sampling_logp_difference/max": 1.6853256225585938, "sampling/importance_sampling_ratio/min": 0.18538405001163483, "sampling/importance_sampling_ratio/mean": 0.9984402060508728, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9484239295125008, "clip_ratio/low_mean": 0.056068320758640766, "clip_ratio/low_min": 0.056068320758640766, "clip_ratio/high_mean": 0.09317009709775448, "clip_ratio/high_max": 0.09317009709775448, "clip_ratio/region_mean": 0.14923841785639524, "reward_total_mean": 0.5273675918579102, "reward_meter_mean": 0.7046166062355042, "reward_meter_std": 0.33246853947639465, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8811008334159851, "reward_repeat_soft_std": 0.04722922295331955, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.5273675918579102, "reward_total_composite_std": 0.08341158181428909} {"timestamp_utc": "2026-04-13T09:48:55Z", "mode": "train", "global_step": 981, "epoch": 0.09854344550477147, "loss": 0.0621, "grad_norm": 12.545612335205078, "learning_rate": 7.030303030303031e-06, "num_tokens": 1730106.0, "completions/mean_length": 52.25, "completions/min_length": 48.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.25, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.4504271447658539, "rewards/meter/std": 0.4088876247406006, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9795470237731934, "rewards/repeat_soft/std": 0.03572678565979004, "rewards/judge_quality/mean": 0.5275000333786011, "rewards/judge_quality/std": 0.18873640894889832, "rewards/total_composite/mean": 0.48322439193725586, "rewards/total_composite/std": 0.10845566540956497, "reward": 0.48322439193725586, "reward_std": 0.10845565795898438, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16551512479782104, "sampling/sampling_logp_difference/max": 3.132415771484375, "sampling/importance_sampling_ratio/min": 0.04361231252551079, "sampling/importance_sampling_ratio/mean": 1.012925386428833, "sampling/importance_sampling_ratio/max": 1.972832441329956, "entropy": 1.3978762775659561, "clip_ratio/low_mean": 0.05751907266676426, "clip_ratio/low_min": 0.05751907266676426, "clip_ratio/high_mean": 0.07019991148263216, "clip_ratio/high_max": 0.07019991148263216, "clip_ratio/region_mean": 0.12771898414939642, "reward_total_mean": 0.48322439193725586, "reward_meter_mean": 0.4504271447658539, "reward_meter_std": 0.4088876247406006, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9795470237731934, "reward_repeat_soft_std": 0.03572678565979004, "reward_judge_quality_mean": 0.5275000333786011, "reward_judge_quality_std": 0.18873640894889832, "reward_total_composite_mean": 0.48322439193725586, "reward_total_composite_std": 0.10845566540956497} {"timestamp_utc": "2026-04-13T09:49:01Z", "mode": "train", "global_step": 982, "epoch": 0.09864389753892516, "loss": 0.007, "grad_norm": 12.460480690002441, "learning_rate": 7.027272727272728e-06, "num_tokens": 1731808.0, "completions/mean_length": 50.75, "completions/min_length": 47.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.75, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.4085719585418701, "rewards/meter/std": 0.3669445812702179, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9877614378929138, "rewards/repeat_soft/std": 0.015099510550498962, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.46311163902282715, "rewards/total_composite/std": 0.10389380156993866, "reward": 0.46311163902282715, "reward_std": 0.10389378666877747, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14943093061447144, "sampling/sampling_logp_difference/max": 1.428863286972046, "sampling/importance_sampling_ratio/min": 0.23958110809326172, "sampling/importance_sampling_ratio/mean": 0.9974090456962585, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0441494435071945, "clip_ratio/low_mean": 0.09914177004247904, "clip_ratio/low_min": 0.09914177004247904, "clip_ratio/high_mean": 0.03319543041288853, "clip_ratio/high_max": 0.03319543041288853, "clip_ratio/region_mean": 0.13233720045536757, "reward_total_mean": 0.46311163902282715, "reward_meter_mean": 0.4085719585418701, "reward_meter_std": 0.3669445812702179, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9877614378929138, "reward_repeat_soft_std": 0.015099510550498962, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.46311163902282715, "reward_total_composite_std": 0.10389380156993866} {"timestamp_utc": "2026-04-13T09:49:08Z", "mode": "train", "global_step": 983, "epoch": 0.09874434957307886, "loss": 0.0131, "grad_norm": 9.582853317260742, "learning_rate": 7.024242424242424e-06, "num_tokens": 1734194.0, "completions/mean_length": 115.25, "completions/min_length": 106.0, "completions/max_length": 122.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.25, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 122.0, "rewards/meter/mean": 0.5660096406936646, "rewards/meter/std": 0.3683083951473236, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9500572681427002, "rewards/repeat_soft/std": 0.04081742838025093, "rewards/judge_quality/mean": 0.9200000166893005, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.6809823513031006, "rewards/total_composite/std": 0.21884137392044067, "reward": 0.6809823513031006, "reward_std": 0.21884137392044067, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.152125746011734, "sampling/sampling_logp_difference/max": 1.720388412475586, "sampling/importance_sampling_ratio/min": 0.17899660766124725, "sampling/importance_sampling_ratio/mean": 1.0238882303237915, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0740713849663734, "clip_ratio/low_mean": 0.10374108701944351, "clip_ratio/low_min": 0.10374108701944351, "clip_ratio/high_mean": 0.05396524164825678, "clip_ratio/high_max": 0.05396524164825678, "clip_ratio/region_mean": 0.1577063286677003, "reward_total_mean": 0.6809823513031006, "reward_meter_mean": 0.5660096406936646, "reward_meter_std": 0.3683083951473236, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9500572681427002, "reward_repeat_soft_std": 0.04081742838025093, "reward_judge_quality_mean": 0.9200000166893005, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.6809823513031006, "reward_total_composite_std": 0.21884137392044067} {"timestamp_utc": "2026-04-13T09:49:14Z", "mode": "train", "global_step": 984, "epoch": 0.09884480160723255, "loss": 0.0445, "grad_norm": 14.900094032287598, "learning_rate": 7.021212121212122e-06, "num_tokens": 1735716.0, "completions/mean_length": 38.25, "completions/min_length": 34.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.25, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.1784151941537857, "rewards/meter/std": 0.21362148225307465, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9586414694786072, "rewards/repeat_soft/std": 0.042823221534490585, "rewards/judge_quality/mean": 0.5387499928474426, "rewards/judge_quality/std": 0.24485784769058228, "rewards/total_composite/mean": 0.41910111904144287, "rewards/total_composite/std": 0.13274075090885162, "reward": 0.41910111904144287, "reward_std": 0.13274075090885162, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14827175438404083, "sampling/sampling_logp_difference/max": 1.7476634979248047, "sampling/importance_sampling_ratio/min": 0.17418043315410614, "sampling/importance_sampling_ratio/mean": 1.0331103801727295, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8540159165859222, "clip_ratio/low_mean": 0.1263057179749012, "clip_ratio/low_min": 0.1263057179749012, "clip_ratio/high_mean": 0.03193196281790733, "clip_ratio/high_max": 0.03193196281790733, "clip_ratio/region_mean": 0.15823768079280853, "reward_total_mean": 0.41910111904144287, "reward_meter_mean": 0.1784151941537857, "reward_meter_std": 0.21362148225307465, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9586414694786072, "reward_repeat_soft_std": 0.042823221534490585, "reward_judge_quality_mean": 0.5387499928474426, "reward_judge_quality_std": 0.24485784769058228, "reward_total_composite_mean": 0.41910111904144287, "reward_total_composite_std": 0.13274075090885162} {"timestamp_utc": "2026-04-13T09:49:20Z", "mode": "train", "global_step": 985, "epoch": 0.09894525364138623, "loss": 0.0334, "grad_norm": 11.730185508728027, "learning_rate": 7.018181818181818e-06, "num_tokens": 1737657.0, "completions/mean_length": 66.625, "completions/min_length": 62.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.625, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.6742357015609741, "rewards/meter/std": 0.3289588689804077, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9603241682052612, "rewards/repeat_soft/std": 0.03282156214118004, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.0645727664232254, "rewards/total_composite/mean": 0.5172199010848999, "rewards/total_composite/std": 0.08268414437770844, "reward": 0.5172199010848999, "reward_std": 0.08268412947654724, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12095610797405243, "sampling/sampling_logp_difference/max": 1.2483692169189453, "sampling/importance_sampling_ratio/min": 0.28697243332862854, "sampling/importance_sampling_ratio/mean": 1.0194545984268188, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9505049139261246, "clip_ratio/low_mean": 0.0459558037109673, "clip_ratio/low_min": 0.0459558037109673, "clip_ratio/high_mean": 0.0553629444912076, "clip_ratio/high_max": 0.0553629444912076, "clip_ratio/region_mean": 0.1013187482021749, "reward_total_mean": 0.5172199010848999, "reward_meter_mean": 0.6742357015609741, "reward_meter_std": 0.3289588689804077, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9603241682052612, "reward_repeat_soft_std": 0.03282156214118004, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.0645727664232254, "reward_total_composite_mean": 0.5172199010848999, "reward_total_composite_std": 0.08268414437770844} {"timestamp_utc": "2026-04-13T09:49:31Z", "mode": "train", "global_step": 986, "epoch": 0.09904570567553993, "loss": -0.1604, "grad_norm": 3.1272151470184326, "learning_rate": 7.015151515151516e-06, "num_tokens": 1739559.0, "completions/mean_length": 146.75, "completions/min_length": 89.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 94.5714340209961, "completions/min_terminated_length": 89.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.4578801393508911, "rewards/meter/std": 0.35649222135543823, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8577362895011902, "rewards/repeat_soft/std": 0.08329711109399796, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.1922610104084015, "rewards/total_composite/mean": 0.40382319688796997, "rewards/total_composite/std": 0.19341367483139038, "reward": 0.40382319688796997, "reward_std": 0.19341367483139038, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13537777960300446, "sampling/sampling_logp_difference/max": 2.4759693145751953, "sampling/importance_sampling_ratio/min": 0.08408144861459732, "sampling/importance_sampling_ratio/mean": 1.0210380554199219, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8224976658821106, "clip_ratio/low_mean": 0.025727187283337116, "clip_ratio/low_min": 0.025727187283337116, "clip_ratio/high_mean": 0.07636873703449965, "clip_ratio/high_max": 0.07636873703449965, "clip_ratio/region_mean": 0.10209592431783676, "reward_total_mean": 0.40382319688796997, "reward_meter_mean": 0.4578801393508911, "reward_meter_std": 0.35649222135543823, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8577362895011902, "reward_repeat_soft_std": 0.08329711109399796, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.1922610104084015, "reward_total_composite_mean": 0.40382319688796997, "reward_total_composite_std": 0.19341367483139038} {"timestamp_utc": "2026-04-13T09:49:38Z", "mode": "train", "global_step": 987, "epoch": 0.09914615770969362, "loss": 0.0662, "grad_norm": 8.873651504516602, "learning_rate": 7.0121212121212126e-06, "num_tokens": 1741599.0, "completions/mean_length": 92.0, "completions/min_length": 84.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.0, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.35385987162590027, "rewards/meter/std": 0.2851884067058563, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8786883354187012, "rewards/repeat_soft/std": 0.10092589259147644, "rewards/judge_quality/mean": 0.7274999618530273, "rewards/judge_quality/std": 0.2549930214881897, "rewards/total_composite/mean": 0.46242237091064453, "rewards/total_composite/std": 0.2586142420768738, "reward": 0.46242237091064453, "reward_std": 0.2586142420768738, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13277406990528107, "sampling/sampling_logp_difference/max": 3.5209875106811523, "sampling/importance_sampling_ratio/min": 0.029570220038294792, "sampling/importance_sampling_ratio/mean": 1.0147207975387573, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.792277567088604, "clip_ratio/low_mean": 0.06794197112321854, "clip_ratio/low_min": 0.06794197112321854, "clip_ratio/high_mean": 0.07332353945821524, "clip_ratio/high_max": 0.07332353945821524, "clip_ratio/region_mean": 0.14126551058143377, "reward_total_mean": 0.46242237091064453, "reward_meter_mean": 0.35385987162590027, "reward_meter_std": 0.2851884067058563, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8786883354187012, "reward_repeat_soft_std": 0.10092589259147644, "reward_judge_quality_mean": 0.7274999618530273, "reward_judge_quality_std": 0.2549930214881897, "reward_total_composite_mean": 0.46242237091064453, "reward_total_composite_std": 0.2586142420768738} {"timestamp_utc": "2026-04-13T09:49:49Z", "mode": "train", "global_step": 988, "epoch": 0.09924660974384732, "loss": -0.1923, "grad_norm": 2.268547296524048, "learning_rate": 7.00909090909091e-06, "num_tokens": 1743645.0, "completions/mean_length": 150.75, "completions/min_length": 87.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 99.14286041259766, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 112.0, "rewards/meter/mean": 0.966325044631958, "rewards/meter/std": 0.019533870741724968, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9042581915855408, "rewards/repeat_soft/std": 0.053632546216249466, "rewards/judge_quality/mean": 0.4112499952316284, "rewards/judge_quality/std": 0.1797965168952942, "rewards/total_composite/mean": 0.5403504967689514, "rewards/total_composite/std": 0.22497345507144928, "reward": 0.5403504967689514, "reward_std": 0.22497345507144928, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12875798344612122, "sampling/sampling_logp_difference/max": 1.6117757558822632, "sampling/importance_sampling_ratio/min": 0.19953297078609467, "sampling/importance_sampling_ratio/mean": 1.023693323135376, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0199998915195465, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.10916881449520588, "clip_ratio/high_max": 0.10916881449520588, "clip_ratio/region_mean": 0.10916881449520588, "reward_total_mean": 0.5403504967689514, "reward_meter_mean": 0.966325044631958, "reward_meter_std": 0.019533870741724968, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9042581915855408, "reward_repeat_soft_std": 0.053632546216249466, "reward_judge_quality_mean": 0.4112499952316284, "reward_judge_quality_std": 0.1797965168952942, "reward_total_composite_mean": 0.5403504967689514, "reward_total_composite_std": 0.22497345507144928} {"timestamp_utc": "2026-04-13T09:49:55Z", "mode": "train", "global_step": 989, "epoch": 0.09934706177800101, "loss": 0.0492, "grad_norm": 15.93040657043457, "learning_rate": 7.006060606060606e-06, "num_tokens": 1745120.0, "completions/mean_length": 35.375, "completions/min_length": 30.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.375, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.4848489761352539, "rewards/meter/std": 0.3328888714313507, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9595526456832886, "rewards/repeat_soft/std": 0.03687829524278641, "rewards/judge_quality/mean": 0.6737500429153442, "rewards/judge_quality/std": 0.263435423374176, "rewards/total_composite/mean": 0.5670564770698547, "rewards/total_composite/std": 0.18828682601451874, "reward": 0.5670564770698547, "reward_std": 0.18828679621219635, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1228448748588562, "sampling/sampling_logp_difference/max": 1.2438445091247559, "sampling/importance_sampling_ratio/min": 0.2882738411426544, "sampling/importance_sampling_ratio/mean": 1.0155507326126099, "sampling/importance_sampling_ratio/max": 1.9814753532409668, "entropy": 0.8090120628476143, "clip_ratio/low_mean": 0.08419212605804205, "clip_ratio/low_min": 0.08419212605804205, "clip_ratio/high_mean": 0.04381816182285547, "clip_ratio/high_max": 0.04381816182285547, "clip_ratio/region_mean": 0.12801028788089752, "reward_total_mean": 0.5670564770698547, "reward_meter_mean": 0.4848489761352539, "reward_meter_std": 0.3328888714313507, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9595526456832886, "reward_repeat_soft_std": 0.03687829524278641, "reward_judge_quality_mean": 0.6737500429153442, "reward_judge_quality_std": 0.263435423374176, "reward_total_composite_mean": 0.5670564770698547, "reward_total_composite_std": 0.18828682601451874} {"timestamp_utc": "2026-04-13T09:50:03Z", "mode": "train", "global_step": 990, "epoch": 0.09944751381215469, "loss": 0.0151, "grad_norm": 8.560992240905762, "learning_rate": 7.0030303030303035e-06, "num_tokens": 1747617.0, "completions/mean_length": 103.125, "completions/min_length": 93.0, "completions/max_length": 124.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 103.125, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.7816253900527954, "rewards/meter/std": 0.2825448513031006, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9555389285087585, "rewards/repeat_soft/std": 0.019325638189911842, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.4619530439376831, "rewards/total_composite/std": 0.24237580597400665, "reward": 0.4619530439376831, "reward_std": 0.24237582087516785, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1434517651796341, "sampling/sampling_logp_difference/max": 1.60699462890625, "sampling/importance_sampling_ratio/min": 0.20048925280570984, "sampling/importance_sampling_ratio/mean": 1.0370699167251587, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.046807236969471, "clip_ratio/low_mean": 0.04783136956393719, "clip_ratio/low_min": 0.04783136956393719, "clip_ratio/high_mean": 0.06494290893897414, "clip_ratio/high_max": 0.06494290893897414, "clip_ratio/region_mean": 0.11277427850291133, "reward_total_mean": 0.4619530439376831, "reward_meter_mean": 0.7816253900527954, "reward_meter_std": 0.2825448513031006, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9555389285087585, "reward_repeat_soft_std": 0.019325638189911842, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.4619530439376831, "reward_total_composite_std": 0.24237580597400665} {"timestamp_utc": "2026-04-13T09:50:09Z", "mode": "train", "global_step": 991, "epoch": 0.09954796584630839, "loss": 0.014, "grad_norm": 19.645828247070312, "learning_rate": 7e-06, "num_tokens": 1748957.0, "completions/mean_length": 25.5, "completions/min_length": 18.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.5, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.7838623523712158, "rewards/meter/std": 0.29824531078338623, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9447510838508606, "rewards/repeat_soft/std": 0.03007800504565239, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.5572038292884827, "rewards/total_composite/std": 0.08105086535215378, "reward": 0.5572038292884827, "reward_std": 0.08105086535215378, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1598208248615265, "sampling/sampling_logp_difference/max": 1.5274133682250977, "sampling/importance_sampling_ratio/min": 0.2170964926481247, "sampling/importance_sampling_ratio/mean": 1.0279059410095215, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.116222970187664, "clip_ratio/low_mean": 0.04640468209981918, "clip_ratio/low_min": 0.04640468209981918, "clip_ratio/high_mean": 0.08125012461096048, "clip_ratio/high_max": 0.08125012461096048, "clip_ratio/region_mean": 0.12765480671077967, "reward_total_mean": 0.5572038292884827, "reward_meter_mean": 0.7838623523712158, "reward_meter_std": 0.29824531078338623, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9447510838508606, "reward_repeat_soft_std": 0.03007800504565239, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.5572038292884827, "reward_total_composite_std": 0.08105086535215378} {"timestamp_utc": "2026-04-13T09:50:16Z", "mode": "train", "global_step": 992, "epoch": 0.09964841788046208, "loss": 0.0397, "grad_norm": 9.281546592712402, "learning_rate": 6.996969696969698e-06, "num_tokens": 1751097.0, "completions/mean_length": 98.5, "completions/min_length": 88.0, "completions/max_length": 111.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.5, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.799793541431427, "rewards/meter/std": 0.3340491056442261, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9555978178977966, "rewards/repeat_soft/std": 0.029661815613508224, "rewards/judge_quality/mean": 0.3687499761581421, "rewards/judge_quality/std": 0.186581090092659, "rewards/total_composite/mean": 0.5163706541061401, "rewards/total_composite/std": 0.14937683939933777, "reward": 0.5163706541061401, "reward_std": 0.14937682449817657, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13694427907466888, "sampling/sampling_logp_difference/max": 3.1758790016174316, "sampling/importance_sampling_ratio/min": 0.041757382452487946, "sampling/importance_sampling_ratio/mean": 1.0145255327224731, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9991964176297188, "clip_ratio/low_mean": 0.05222881771624088, "clip_ratio/low_min": 0.05222881771624088, "clip_ratio/high_mean": 0.05285879969596863, "clip_ratio/high_max": 0.05285879969596863, "clip_ratio/region_mean": 0.10508761741220951, "reward_total_mean": 0.5163706541061401, "reward_meter_mean": 0.799793541431427, "reward_meter_std": 0.3340491056442261, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9555978178977966, "reward_repeat_soft_std": 0.029661815613508224, "reward_judge_quality_mean": 0.3687499761581421, "reward_judge_quality_std": 0.186581090092659, "reward_total_composite_mean": 0.5163706541061401, "reward_total_composite_std": 0.14937683939933777} {"timestamp_utc": "2026-04-13T09:50:22Z", "mode": "train", "global_step": 993, "epoch": 0.09974886991461578, "loss": 0.0773, "grad_norm": 35.79539489746094, "learning_rate": 6.993939393939394e-06, "num_tokens": 1752483.0, "completions/mean_length": 20.25, "completions/min_length": 17.0, "completions/max_length": 25.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 20.25, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 25.0, "rewards/meter/mean": 0.6874057054519653, "rewards/meter/std": 0.4219528138637543, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9263834953308105, "rewards/repeat_soft/std": 0.0961892157793045, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.5270617008209229, "rewards/total_composite/std": 0.11126212030649185, "reward": 0.5270617008209229, "reward_std": 0.11126213520765305, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.151984304189682, "sampling/sampling_logp_difference/max": 1.055746078491211, "sampling/importance_sampling_ratio/min": 0.34793272614479065, "sampling/importance_sampling_ratio/mean": 1.059171199798584, "sampling/importance_sampling_ratio/max": 1.8717169761657715, "entropy": 1.2759125232696533, "clip_ratio/low_mean": 0.0470467833802104, "clip_ratio/low_min": 0.0470467833802104, "clip_ratio/high_mean": 0.11265883315354586, "clip_ratio/high_max": 0.11265883315354586, "clip_ratio/region_mean": 0.15970561653375626, "reward_total_mean": 0.5270617008209229, "reward_meter_mean": 0.6874057054519653, "reward_meter_std": 0.4219528138637543, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9263834953308105, "reward_repeat_soft_std": 0.0961892157793045, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.5270617008209229, "reward_total_composite_std": 0.11126212030649185} {"timestamp_utc": "2026-04-13T09:50:34Z", "mode": "train", "global_step": 994, "epoch": 0.09984932194876946, "loss": -0.1729, "grad_norm": 2.7367634773254395, "learning_rate": 6.990909090909092e-06, "num_tokens": 1754302.0, "completions/mean_length": 129.375, "completions/min_length": 66.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 74.71428680419922, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.7008242607116699, "rewards/meter/std": 0.3392601013183594, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9867971539497375, "rewards/repeat_soft/std": 0.004942748695611954, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.13593590259552002, "rewards/total_composite/mean": 0.4817453920841217, "rewards/total_composite/std": 0.2017514705657959, "reward": 0.4817453920841217, "reward_std": 0.2017514556646347, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1783118098974228, "sampling/sampling_logp_difference/max": 1.7253952026367188, "sampling/importance_sampling_ratio/min": 0.17810265719890594, "sampling/importance_sampling_ratio/mean": 1.037008285522461, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2566193640232086, "clip_ratio/low_mean": 0.02651515230536461, "clip_ratio/low_min": 0.02651515230536461, "clip_ratio/high_mean": 0.09103912860155106, "clip_ratio/high_max": 0.09103912860155106, "clip_ratio/region_mean": 0.11755428090691566, "reward_total_mean": 0.4817453920841217, "reward_meter_mean": 0.7008242607116699, "reward_meter_std": 0.3392601013183594, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9867971539497375, "reward_repeat_soft_std": 0.004942748695611954, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.13593590259552002, "reward_total_composite_mean": 0.4817453920841217, "reward_total_composite_std": 0.2017514705657959} {"timestamp_utc": "2026-04-13T09:50:45Z", "mode": "train", "global_step": 995, "epoch": 0.09994977398292315, "loss": -0.0748, "grad_norm": 2.902416944503784, "learning_rate": 6.987878787878788e-06, "num_tokens": 1755856.0, "completions/mean_length": 94.25, "completions/min_length": 30.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 34.57143020629883, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.604584813117981, "rewards/meter/std": 0.32237282395362854, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9927259683609009, "rewards/repeat_soft/std": 0.01288491953164339, "rewards/judge_quality/mean": 0.627500057220459, "rewards/judge_quality/std": 0.3366537392139435, "rewards/total_composite/mean": 0.5558649301528931, "rewards/total_composite/std": 0.28423136472702026, "reward": 0.5558649301528931, "reward_std": 0.28423136472702026, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11629508435726166, "sampling/sampling_logp_difference/max": 1.8717248439788818, "sampling/importance_sampling_ratio/min": 0.15385805070400238, "sampling/importance_sampling_ratio/mean": 0.9965556263923645, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4548973888158798, "clip_ratio/low_mean": 0.04940191563218832, "clip_ratio/low_min": 0.04940191563218832, "clip_ratio/high_mean": 0.022852607304230332, "clip_ratio/high_max": 0.022852607304230332, "clip_ratio/region_mean": 0.07225452293641865, "reward_total_mean": 0.5558649301528931, "reward_meter_mean": 0.604584813117981, "reward_meter_std": 0.32237282395362854, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9927259683609009, "reward_repeat_soft_std": 0.01288491953164339, "reward_judge_quality_mean": 0.627500057220459, "reward_judge_quality_std": 0.3366537392139435, "reward_total_composite_mean": 0.5558649301528931, "reward_total_composite_std": 0.28423136472702026} {"timestamp_utc": "2026-04-13T09:50:51Z", "mode": "train", "global_step": 996, "epoch": 0.10005022601707685, "loss": 0.0452, "grad_norm": 15.451910018920898, "learning_rate": 6.984848484848485e-06, "num_tokens": 1757382.0, "completions/mean_length": 27.75, "completions/min_length": 24.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.75, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.6767911314964294, "rewards/meter/std": 0.40244999527931213, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.933674693107605, "rewards/repeat_soft/std": 0.036771856248378754, "rewards/judge_quality/mean": 0.6150000095367432, "rewards/judge_quality/std": 0.25286927819252014, "rewards/total_composite/mean": 0.6221144199371338, "rewards/total_composite/std": 0.19765277206897736, "reward": 0.6221144199371338, "reward_std": 0.19765277206897736, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1169843077659607, "sampling/sampling_logp_difference/max": 1.4371109008789062, "sampling/importance_sampling_ratio/min": 0.2376132607460022, "sampling/importance_sampling_ratio/mean": 1.033090591430664, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9760359674692154, "clip_ratio/low_mean": 0.03587464150041342, "clip_ratio/low_min": 0.03587464150041342, "clip_ratio/high_mean": 0.03456959826871753, "clip_ratio/high_max": 0.03456959826871753, "clip_ratio/region_mean": 0.07044423976913095, "reward_total_mean": 0.6221144199371338, "reward_meter_mean": 0.6767911314964294, "reward_meter_std": 0.40244999527931213, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.933674693107605, "reward_repeat_soft_std": 0.036771856248378754, "reward_judge_quality_mean": 0.6150000095367432, "reward_judge_quality_std": 0.25286927819252014, "reward_total_composite_mean": 0.6221144199371338, "reward_total_composite_std": 0.19765277206897736} {"timestamp_utc": "2026-04-13T09:50:57Z", "mode": "train", "global_step": 997, "epoch": 0.10015067805123054, "loss": -0.0252, "grad_norm": 14.92132568359375, "learning_rate": 6.981818181818183e-06, "num_tokens": 1758955.0, "completions/mean_length": 46.625, "completions/min_length": 37.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.625, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.7483348846435547, "rewards/meter/std": 0.3284272849559784, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9753111004829407, "rewards/repeat_soft/std": 0.03309660404920578, "rewards/judge_quality/mean": 0.4762499928474426, "rewards/judge_quality/std": 0.11160357296466827, "rewards/total_composite/mean": 0.5082617998123169, "rewards/total_composite/std": 0.24537086486816406, "reward": 0.5082617998123169, "reward_std": 0.24537083506584167, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16312509775161743, "sampling/sampling_logp_difference/max": 1.5516786575317383, "sampling/importance_sampling_ratio/min": 0.21189197897911072, "sampling/importance_sampling_ratio/mean": 1.0018374919891357, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.112735353410244, "clip_ratio/low_mean": 0.07301124837249517, "clip_ratio/low_min": 0.07301124837249517, "clip_ratio/high_mean": 0.0901602627709508, "clip_ratio/high_max": 0.0901602627709508, "clip_ratio/region_mean": 0.16317151114344597, "reward_total_mean": 0.5082617998123169, "reward_meter_mean": 0.7483348846435547, "reward_meter_std": 0.3284272849559784, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9753111004829407, "reward_repeat_soft_std": 0.03309660404920578, "reward_judge_quality_mean": 0.4762499928474426, "reward_judge_quality_std": 0.11160357296466827, "reward_total_composite_mean": 0.5082617998123169, "reward_total_composite_std": 0.24537086486816406} {"timestamp_utc": "2026-04-13T09:51:03Z", "mode": "train", "global_step": 998, "epoch": 0.10025113008538424, "loss": 0.0716, "grad_norm": 10.448753356933594, "learning_rate": 6.978787878787879e-06, "num_tokens": 1761004.0, "completions/mean_length": 86.125, "completions/min_length": 71.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.125, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.6371596455574036, "rewards/meter/std": 0.33203747868537903, "rewards/count_adherence/mean": 0.7749999761581421, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.894789457321167, "rewards/repeat_soft/std": 0.037303801625967026, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.46316301822662354, "rewards/total_composite/std": 0.09780687838792801, "reward": 0.46316301822662354, "reward_std": 0.09780686348676682, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15020348131656647, "sampling/sampling_logp_difference/max": 2.2706642150878906, "sampling/importance_sampling_ratio/min": 0.10324358195066452, "sampling/importance_sampling_ratio/mean": 1.0024322271347046, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0162644013762474, "clip_ratio/low_mean": 0.04118903633207083, "clip_ratio/low_min": 0.04118903633207083, "clip_ratio/high_mean": 0.09987357072532177, "clip_ratio/high_max": 0.09987357072532177, "clip_ratio/region_mean": 0.1410626070573926, "reward_total_mean": 0.46316301822662354, "reward_meter_mean": 0.6371596455574036, "reward_meter_std": 0.33203747868537903, "reward_count_adherence_mean": 0.7749999761581421, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.894789457321167, "reward_repeat_soft_std": 0.037303801625967026, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.46316301822662354, "reward_total_composite_std": 0.09780687838792801} {"timestamp_utc": "2026-04-13T09:51:09Z", "mode": "train", "global_step": 999, "epoch": 0.10035158211953792, "loss": 0.0327, "grad_norm": 12.499972343444824, "learning_rate": 6.975757575757577e-06, "num_tokens": 1762845.0, "completions/mean_length": 56.125, "completions/min_length": 49.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.125, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.8660759329795837, "rewards/meter/std": 0.3027213215827942, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9728718996047974, "rewards/repeat_soft/std": 0.0173123087733984, "rewards/judge_quality/mean": 0.44875001907348633, "rewards/judge_quality/std": 0.21256513893604279, "rewards/total_composite/mean": 0.6023823618888855, "rewards/total_composite/std": 0.16437385976314545, "reward": 0.6023823618888855, "reward_std": 0.16437385976314545, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15853369235992432, "sampling/sampling_logp_difference/max": 1.2867321968078613, "sampling/importance_sampling_ratio/min": 0.2761717736721039, "sampling/importance_sampling_ratio/mean": 1.0485488176345825, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.425296112895012, "clip_ratio/low_mean": 0.03546767123043537, "clip_ratio/low_min": 0.03546767123043537, "clip_ratio/high_mean": 0.09489201102405787, "clip_ratio/high_max": 0.09489201102405787, "clip_ratio/region_mean": 0.13035968225449324, "reward_total_mean": 0.6023823618888855, "reward_meter_mean": 0.8660759329795837, "reward_meter_std": 0.3027213215827942, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9728718996047974, "reward_repeat_soft_std": 0.0173123087733984, "reward_judge_quality_mean": 0.44875001907348633, "reward_judge_quality_std": 0.21256513893604279, "reward_total_composite_mean": 0.6023823618888855, "reward_total_composite_std": 0.16437385976314545} {"timestamp_utc": "2026-04-13T09:51:16Z", "mode": "train", "global_step": 1000, "epoch": 0.10045203415369161, "loss": 0.0485, "grad_norm": 9.855438232421875, "learning_rate": 6.9727272727272735e-06, "num_tokens": 1764638.0, "completions/mean_length": 76.125, "completions/min_length": 67.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.125, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.2827584743499756, "rewards/meter/std": 0.3035575747489929, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9801421165466309, "rewards/repeat_soft/std": 0.015814922749996185, "rewards/judge_quality/mean": 0.5237500071525574, "rewards/judge_quality/std": 0.19078317284584045, "rewards/total_composite/mean": 0.4346918761730194, "rewards/total_composite/std": 0.08062057942152023, "reward": 0.4346918761730194, "reward_std": 0.08062057942152023, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15195401012897491, "sampling/sampling_logp_difference/max": 2.9526617527008057, "sampling/importance_sampling_ratio/min": 0.052200574427843094, "sampling/importance_sampling_ratio/mean": 1.0094891786575317, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0911433175206184, "clip_ratio/low_mean": 0.07451982609927654, "clip_ratio/low_min": 0.07451982609927654, "clip_ratio/high_mean": 0.049187893979251385, "clip_ratio/high_max": 0.049187893979251385, "clip_ratio/region_mean": 0.12370772007852793, "reward_total_mean": 0.4346918761730194, "reward_meter_mean": 0.2827584743499756, "reward_meter_std": 0.3035575747489929, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9801421165466309, "reward_repeat_soft_std": 0.015814922749996185, "reward_judge_quality_mean": 0.5237500071525574, "reward_judge_quality_std": 0.19078317284584045, "reward_total_composite_mean": 0.4346918761730194, "reward_total_composite_std": 0.08062057942152023} {"timestamp_utc": "2026-04-13T09:52:01Z", "mode": "eval", "global_step": 1000, "epoch": 0.10045203415369161, "eval_loss": NaN, "eval_runtime": 44.8301, "eval_samples_per_second": 1.785, "eval_steps_per_second": 0.223, "eval_num_tokens": 1764638.0, "eval_completions/mean_length": 70.9125, "eval_completions/min_length": 35.5, "eval_completions/max_length": 141.0, "eval_completions/clipped_ratio": 0.0125, "eval_completions/mean_terminated_length": 65.25357170104981, "eval_completions/min_terminated_length": 35.5, "eval_completions/max_terminated_length": 98.6, "eval_rewards/meter/mean": 0.6456027030944824, "eval_rewards/meter/std": 0.3476584076881409, "eval_rewards/count_adherence/mean": 0.8720833361148834, "eval_rewards/count_adherence/std": 0.158842521160841, "eval_rewards/hard_gate/mean": 0.975, "eval_rewards/hard_gate/std": 0.07071067690849304, "eval_rewards/repeat_soft/mean": 0.9484191417694092, "eval_rewards/repeat_soft/std": 0.04719752427190542, "eval_rewards/judge_quality/mean": 0.46599999666213987, "eval_rewards/judge_quality/std": 0.17726094285026192, "eval_rewards/total_composite/mean": 0.5045325458049774, "eval_rewards/total_composite/std": 0.15694166645407676, "eval_reward": 0.5045325458049774, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.08031205013394356, "eval_sampling/sampling_logp_difference/max": 1.203929328918457, "eval_sampling/importance_sampling_ratio/min": 0.3103459820151329, "eval_sampling/importance_sampling_ratio/mean": 1.0168931245803834, "eval_sampling/importance_sampling_ratio/max": 1.449403500556946, "eval_entropy": 0.8986621379852295, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5045325458049774, "eval_reward_meter_mean": 0.6456027030944824, "eval_reward_meter_std": 0.3476584076881409, "eval_reward_count_adherence_mean": 0.8720833361148834, "eval_reward_count_adherence_std": 0.158842521160841, "eval_reward_hard_gate_mean": 0.975, "eval_reward_hard_gate_std": 0.07071067690849304, "eval_reward_repeat_soft_mean": 0.9484191417694092, "eval_reward_repeat_soft_std": 0.04719752427190542, "eval_reward_judge_quality_mean": 0.46599999666213987, "eval_reward_judge_quality_std": 0.17726094285026192, "eval_reward_total_composite_mean": 0.5045325458049774, "eval_reward_total_composite_std": 0.15694166645407676} {"timestamp_utc": "2026-04-13T09:52:11Z", "mode": "train", "global_step": 1001, "epoch": 0.1005524861878453, "loss": 0.0217, "grad_norm": 9.642687797546387, "learning_rate": 6.969696969696971e-06, "num_tokens": 1766852.0, "completions/mean_length": 82.75, "completions/min_length": 72.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.75, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.6661903262138367, "rewards/meter/std": 0.30564743280410767, "rewards/count_adherence/mean": 0.7250000238418579, "rewards/count_adherence/std": 0.1035098284482956, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8967072367668152, "rewards/repeat_soft/std": 0.04659102112054825, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.4313625395298004, "rewards/total_composite/std": 0.09541169553995132, "reward": 0.4313625395298004, "reward_std": 0.09541170299053192, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1348487138748169, "sampling/sampling_logp_difference/max": 2.945591926574707, "sampling/importance_sampling_ratio/min": 0.05257093533873558, "sampling/importance_sampling_ratio/mean": 1.0046889781951904, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.808423787355423, "clip_ratio/low_mean": 0.04772999323904514, "clip_ratio/low_min": 0.04772999323904514, "clip_ratio/high_mean": 0.09033445827662945, "clip_ratio/high_max": 0.09033445827662945, "clip_ratio/region_mean": 0.1380644515156746, "reward_total_mean": 0.4313625395298004, "reward_meter_mean": 0.6661903262138367, "reward_meter_std": 0.30564743280410767, "reward_count_adherence_mean": 0.7250000238418579, "reward_count_adherence_std": 0.1035098284482956, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8967072367668152, "reward_repeat_soft_std": 0.04659102112054825, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.4313625395298004, "reward_total_composite_std": 0.09541169553995132} {"timestamp_utc": "2026-04-13T09:52:17Z", "mode": "train", "global_step": 1002, "epoch": 0.100652938221999, "loss": -0.295, "grad_norm": 5.131814002990723, "learning_rate": 6.966666666666667e-06, "num_tokens": 1768564.0, "completions/mean_length": 62.0, "completions/min_length": 10.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.0, "completions/min_terminated_length": 10.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.849102258682251, "rewards/meter/std": 0.34443584084510803, "rewards/count_adherence/mean": 0.7916666865348816, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9541321396827698, "rewards/repeat_soft/std": 0.038377951830625534, "rewards/judge_quality/mean": 0.38624998927116394, "rewards/judge_quality/std": 0.09545940905809402, "rewards/total_composite/mean": 0.514508068561554, "rewards/total_composite/std": 0.21034754812717438, "reward": 0.514508068561554, "reward_std": 0.21034754812717438, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13236641883850098, "sampling/sampling_logp_difference/max": 2.604266405105591, "sampling/importance_sampling_ratio/min": 0.07395736873149872, "sampling/importance_sampling_ratio/mean": 0.992257833480835, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1153065040707588, "clip_ratio/low_mean": 0.03750000149011612, "clip_ratio/low_min": 0.03750000149011612, "clip_ratio/high_mean": 0.11825434910133481, "clip_ratio/high_max": 0.11825434910133481, "clip_ratio/region_mean": 0.15575435059145093, "reward_total_mean": 0.514508068561554, "reward_meter_mean": 0.849102258682251, "reward_meter_std": 0.34443584084510803, "reward_count_adherence_mean": 0.7916666865348816, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9541321396827698, "reward_repeat_soft_std": 0.038377951830625534, "reward_judge_quality_mean": 0.38624998927116394, "reward_judge_quality_std": 0.09545940905809402, "reward_total_composite_mean": 0.514508068561554, "reward_total_composite_std": 0.21034754812717438} {"timestamp_utc": "2026-04-13T09:52:24Z", "mode": "train", "global_step": 1003, "epoch": 0.10075339025615268, "loss": 0.0668, "grad_norm": 16.053508758544922, "learning_rate": 6.963636363636364e-06, "num_tokens": 1770278.0, "completions/mean_length": 35.25, "completions/min_length": 30.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.25, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.4937446713447571, "rewards/meter/std": 0.2696358263492584, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8978832960128784, "rewards/repeat_soft/std": 0.08877812325954437, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.48789727687835693, "rewards/total_composite/std": 0.0975428968667984, "reward": 0.48789727687835693, "reward_std": 0.0975428894162178, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1792350709438324, "sampling/sampling_logp_difference/max": 1.2945919036865234, "sampling/importance_sampling_ratio/min": 0.274009644985199, "sampling/importance_sampling_ratio/mean": 1.033050775527954, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6600244417786598, "clip_ratio/low_mean": 0.08472664840519428, "clip_ratio/low_min": 0.08472664840519428, "clip_ratio/high_mean": 0.07087546866387129, "clip_ratio/high_max": 0.07087546866387129, "clip_ratio/region_mean": 0.15560211706906557, "reward_total_mean": 0.48789727687835693, "reward_meter_mean": 0.4937446713447571, "reward_meter_std": 0.2696358263492584, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8978832960128784, "reward_repeat_soft_std": 0.08877812325954437, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.48789727687835693, "reward_total_composite_std": 0.0975428968667984} {"timestamp_utc": "2026-04-13T09:52:35Z", "mode": "train", "global_step": 1004, "epoch": 0.10085384229030638, "loss": -0.095, "grad_norm": 3.897873640060425, "learning_rate": 6.960606060606061e-06, "num_tokens": 1772071.0, "completions/mean_length": 129.125, "completions/min_length": 53.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 74.42857360839844, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.47224485874176025, "rewards/meter/std": 0.3324452340602875, "rewards/count_adherence/mean": 0.71875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9462336301803589, "rewards/repeat_soft/std": 0.05075499042868614, "rewards/judge_quality/mean": 0.45250001549720764, "rewards/judge_quality/std": 0.25949129462242126, "rewards/total_composite/mean": 0.4135204553604126, "rewards/total_composite/std": 0.21175077557563782, "reward": 0.4135204553604126, "reward_std": 0.21175077557563782, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14971968531608582, "sampling/sampling_logp_difference/max": 2.313939094543457, "sampling/importance_sampling_ratio/min": 0.09887102246284485, "sampling/importance_sampling_ratio/mean": 1.0206713676452637, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9123754799365997, "clip_ratio/low_mean": 0.026376837864518166, "clip_ratio/low_min": 0.026376837864518166, "clip_ratio/high_mean": 0.07666672486811876, "clip_ratio/high_max": 0.07666672486811876, "clip_ratio/region_mean": 0.10304356273263693, "reward_total_mean": 0.4135204553604126, "reward_meter_mean": 0.47224485874176025, "reward_meter_std": 0.3324452340602875, "reward_count_adherence_mean": 0.71875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9462336301803589, "reward_repeat_soft_std": 0.05075499042868614, "reward_judge_quality_mean": 0.45250001549720764, "reward_judge_quality_std": 0.25949129462242126, "reward_total_composite_mean": 0.4135204553604126, "reward_total_composite_std": 0.21175077557563782} {"timestamp_utc": "2026-04-13T09:52:41Z", "mode": "train", "global_step": 1005, "epoch": 0.10095429432446007, "loss": 0.0251, "grad_norm": 23.581077575683594, "learning_rate": 6.957575757575759e-06, "num_tokens": 1773484.0, "completions/mean_length": 21.625, "completions/min_length": 20.0, "completions/max_length": 24.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.625, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 24.0, "rewards/meter/mean": 0.8936288356781006, "rewards/meter/std": 0.2433493435382843, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9524593949317932, "rewards/repeat_soft/std": 0.02408791147172451, "rewards/judge_quality/mean": 0.4012500047683716, "rewards/judge_quality/std": 0.10260012745857239, "rewards/total_composite/mean": 0.5899405479431152, "rewards/total_composite/std": 0.09260963648557663, "reward": 0.5899405479431152, "reward_std": 0.09260963648557663, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1577124148607254, "sampling/sampling_logp_difference/max": 2.8035335540771484, "sampling/importance_sampling_ratio/min": 0.0605955645442009, "sampling/importance_sampling_ratio/mean": 1.0209708213806152, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7599133998155594, "clip_ratio/low_mean": 0.021739130839705467, "clip_ratio/low_min": 0.021739130839705467, "clip_ratio/high_mean": 0.07247023982927203, "clip_ratio/high_max": 0.07247023982927203, "clip_ratio/region_mean": 0.0942093706689775, "reward_total_mean": 0.5899405479431152, "reward_meter_mean": 0.8936288356781006, "reward_meter_std": 0.2433493435382843, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9524593949317932, "reward_repeat_soft_std": 0.02408791147172451, "reward_judge_quality_mean": 0.4012500047683716, "reward_judge_quality_std": 0.10260012745857239, "reward_total_composite_mean": 0.5899405479431152, "reward_total_composite_std": 0.09260963648557663} {"timestamp_utc": "2026-04-13T09:52:47Z", "mode": "train", "global_step": 1006, "epoch": 0.10105474635861376, "loss": -0.0363, "grad_norm": 12.378266334533691, "learning_rate": 6.954545454545455e-06, "num_tokens": 1775445.0, "completions/mean_length": 73.125, "completions/min_length": 67.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.125, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.24075129628181458, "rewards/meter/std": 0.26709258556365967, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9653311371803284, "rewards/repeat_soft/std": 0.016731778159737587, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.41278842091560364, "rewards/total_composite/std": 0.07317467033863068, "reward": 0.41278842091560364, "reward_std": 0.07317467033863068, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14169339835643768, "sampling/sampling_logp_difference/max": 3.1516613960266113, "sampling/importance_sampling_ratio/min": 0.04278099164366722, "sampling/importance_sampling_ratio/mean": 1.0222159624099731, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7837047949433327, "clip_ratio/low_mean": 0.07796356175094843, "clip_ratio/low_min": 0.07796356175094843, "clip_ratio/high_mean": 0.03137254994362593, "clip_ratio/high_max": 0.03137254994362593, "clip_ratio/region_mean": 0.10933611169457436, "reward_total_mean": 0.41278842091560364, "reward_meter_mean": 0.24075129628181458, "reward_meter_std": 0.26709258556365967, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9653311371803284, "reward_repeat_soft_std": 0.016731778159737587, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.41278842091560364, "reward_total_composite_std": 0.07317467033863068} {"timestamp_utc": "2026-04-13T09:52:53Z", "mode": "train", "global_step": 1007, "epoch": 0.10115519839276746, "loss": 0.0718, "grad_norm": 14.407511711120605, "learning_rate": 6.951515151515153e-06, "num_tokens": 1777019.0, "completions/mean_length": 42.75, "completions/min_length": 37.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.75, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.9675606489181519, "rewards/meter/std": 0.031776558607816696, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9352800250053406, "rewards/repeat_soft/std": 0.09628202021121979, "rewards/judge_quality/mean": 0.7325000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.8021653890609741, "rewards/total_composite/std": 0.1718769669532776, "reward": 0.8021653890609741, "reward_std": 0.1718769520521164, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16105450689792633, "sampling/sampling_logp_difference/max": 1.6512861251831055, "sampling/importance_sampling_ratio/min": 0.19180306792259216, "sampling/importance_sampling_ratio/mean": 1.0006808042526245, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1489304453134537, "clip_ratio/low_mean": 0.051672996021807194, "clip_ratio/low_min": 0.051672996021807194, "clip_ratio/high_mean": 0.10751363914459944, "clip_ratio/high_max": 0.10751363914459944, "clip_ratio/region_mean": 0.15918663516640663, "reward_total_mean": 0.8021653890609741, "reward_meter_mean": 0.9675606489181519, "reward_meter_std": 0.031776558607816696, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9352800250053406, "reward_repeat_soft_std": 0.09628202021121979, "reward_judge_quality_mean": 0.7325000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.8021653890609741, "reward_total_composite_std": 0.1718769669532776} {"timestamp_utc": "2026-04-13T09:52:59Z", "mode": "train", "global_step": 1008, "epoch": 0.10125565042692114, "loss": 0.0522, "grad_norm": 9.007136344909668, "learning_rate": 6.948484848484849e-06, "num_tokens": 1778936.0, "completions/mean_length": 63.625, "completions/min_length": 53.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.625, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.79938805103302, "rewards/meter/std": 0.3317202031612396, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9361878037452698, "rewards/repeat_soft/std": 0.08320317417383194, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.6969414949417114, "rewards/total_composite/std": 0.1936228722333908, "reward": 0.6969414949417114, "reward_std": 0.19362285733222961, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15141348540782928, "sampling/sampling_logp_difference/max": 1.6684417724609375, "sampling/importance_sampling_ratio/min": 0.18854062259197235, "sampling/importance_sampling_ratio/mean": 1.0258336067199707, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2222661450505257, "clip_ratio/low_mean": 0.08359825983643532, "clip_ratio/low_min": 0.08359825983643532, "clip_ratio/high_mean": 0.054921312257647514, "clip_ratio/high_max": 0.054921312257647514, "clip_ratio/region_mean": 0.13851957209408283, "reward_total_mean": 0.6969414949417114, "reward_meter_mean": 0.79938805103302, "reward_meter_std": 0.3317202031612396, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9361878037452698, "reward_repeat_soft_std": 0.08320317417383194, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.6969414949417114, "reward_total_composite_std": 0.1936228722333908} {"timestamp_utc": "2026-04-13T09:53:06Z", "mode": "train", "global_step": 1009, "epoch": 0.10135610246107483, "loss": -0.0071, "grad_norm": 12.920982360839844, "learning_rate": 6.945454545454546e-06, "num_tokens": 1780722.0, "completions/mean_length": 57.25, "completions/min_length": 45.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.25, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.6258684396743774, "rewards/meter/std": 0.3698013424873352, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9881253242492676, "rewards/repeat_soft/std": 0.008947896771132946, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.10507651418447495, "rewards/total_composite/mean": 0.5284538865089417, "rewards/total_composite/std": 0.09727796167135239, "reward": 0.5284538865089417, "reward_std": 0.09727796912193298, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16133199632167816, "sampling/sampling_logp_difference/max": 1.516340732574463, "sampling/importance_sampling_ratio/min": 0.2195136696100235, "sampling/importance_sampling_ratio/mean": 1.0304008722305298, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3624932616949081, "clip_ratio/low_mean": 0.037349299527704716, "clip_ratio/low_min": 0.037349299527704716, "clip_ratio/high_mean": 0.11407075077295303, "clip_ratio/high_max": 0.11407075077295303, "clip_ratio/region_mean": 0.15142005030065775, "reward_total_mean": 0.5284538865089417, "reward_meter_mean": 0.6258684396743774, "reward_meter_std": 0.3698013424873352, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9881253242492676, "reward_repeat_soft_std": 0.008947896771132946, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.10507651418447495, "reward_total_composite_mean": 0.5284538865089417, "reward_total_composite_std": 0.09727796167135239} {"timestamp_utc": "2026-04-13T09:53:13Z", "mode": "train", "global_step": 1010, "epoch": 0.10145655449522853, "loss": 0.0691, "grad_norm": 8.525898933410645, "learning_rate": 6.942424242424243e-06, "num_tokens": 1783280.0, "completions/mean_length": 107.75, "completions/min_length": 96.0, "completions/max_length": 127.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.75, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 127.0, "rewards/meter/mean": 0.7836625576019287, "rewards/meter/std": 0.23392094671726227, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9279478788375854, "rewards/repeat_soft/std": 0.024648524820804596, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.46836158633232117, "rewards/total_composite/std": 0.09341593831777573, "reward": 0.46836158633232117, "reward_std": 0.09341592341661453, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16590675711631775, "sampling/sampling_logp_difference/max": 1.8753886222839355, "sampling/importance_sampling_ratio/min": 0.1532953828573227, "sampling/importance_sampling_ratio/mean": 1.0183576345443726, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.105267234146595, "clip_ratio/low_mean": 0.04774219263345003, "clip_ratio/low_min": 0.04774219263345003, "clip_ratio/high_mean": 0.09435775689780712, "clip_ratio/high_max": 0.09435775689780712, "clip_ratio/region_mean": 0.14209994953125715, "reward_total_mean": 0.46836158633232117, "reward_meter_mean": 0.7836625576019287, "reward_meter_std": 0.23392094671726227, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9279478788375854, "reward_repeat_soft_std": 0.024648524820804596, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.46836158633232117, "reward_total_composite_std": 0.09341593831777573} {"timestamp_utc": "2026-04-13T09:53:20Z", "mode": "train", "global_step": 1011, "epoch": 0.10155700652938222, "loss": 0.0427, "grad_norm": 12.076990127563477, "learning_rate": 6.93939393939394e-06, "num_tokens": 1785403.0, "completions/mean_length": 69.375, "completions/min_length": 65.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.375, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9308876395225525, "rewards/meter/std": 0.176197811961174, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9217841625213623, "rewards/repeat_soft/std": 0.0469188392162323, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.5286530256271362, "rewards/total_composite/std": 0.05541308969259262, "reward": 0.5286530256271362, "reward_std": 0.05541309341788292, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12863069772720337, "sampling/sampling_logp_difference/max": 1.9285722970962524, "sampling/importance_sampling_ratio/min": 0.14535556733608246, "sampling/importance_sampling_ratio/mean": 1.018244743347168, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9357447624206543, "clip_ratio/low_mean": 0.0390593558549881, "clip_ratio/low_min": 0.0390593558549881, "clip_ratio/high_mean": 0.07747931219637394, "clip_ratio/high_max": 0.07747931219637394, "clip_ratio/region_mean": 0.11653866805136204, "reward_total_mean": 0.5286530256271362, "reward_meter_mean": 0.9308876395225525, "reward_meter_std": 0.176197811961174, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9217841625213623, "reward_repeat_soft_std": 0.0469188392162323, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.5286530256271362, "reward_total_composite_std": 0.05541308969259262} {"timestamp_utc": "2026-04-13T09:53:26Z", "mode": "train", "global_step": 1012, "epoch": 0.10165745856353592, "loss": -0.0774, "grad_norm": 15.02917194366455, "learning_rate": 6.936363636363636e-06, "num_tokens": 1786934.0, "completions/mean_length": 40.375, "completions/min_length": 32.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.375, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.7910438179969788, "rewards/meter/std": 0.3508176803588867, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.942015528678894, "rewards/repeat_soft/std": 0.054328788071870804, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.5620580911636353, "rewards/total_composite/std": 0.09672543406486511, "reward": 0.5620580911636353, "reward_std": 0.09672542661428452, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1535942405462265, "sampling/sampling_logp_difference/max": 1.341993808746338, "sampling/importance_sampling_ratio/min": 0.2613241374492645, "sampling/importance_sampling_ratio/mean": 1.0322272777557373, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2767742425203323, "clip_ratio/low_mean": 0.022569444496184587, "clip_ratio/low_min": 0.022569444496184587, "clip_ratio/high_mean": 0.10155987646430731, "clip_ratio/high_max": 0.10155987646430731, "clip_ratio/region_mean": 0.1241293209604919, "reward_total_mean": 0.5620580911636353, "reward_meter_mean": 0.7910438179969788, "reward_meter_std": 0.3508176803588867, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.942015528678894, "reward_repeat_soft_std": 0.054328788071870804, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.5620580911636353, "reward_total_composite_std": 0.09672543406486511} {"timestamp_utc": "2026-04-13T09:53:33Z", "mode": "train", "global_step": 1013, "epoch": 0.1017579105976896, "loss": -0.0606, "grad_norm": 10.494180679321289, "learning_rate": 6.9333333333333344e-06, "num_tokens": 1789010.0, "completions/mean_length": 86.5, "completions/min_length": 68.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.5, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.9619123935699463, "rewards/meter/std": 0.02976352907717228, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9517722725868225, "rewards/repeat_soft/std": 0.04128618910908699, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.5409913063049316, "rewards/total_composite/std": 0.231267049908638, "reward": 0.5409913063049316, "reward_std": 0.2312670350074768, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14850583672523499, "sampling/sampling_logp_difference/max": 5.278156280517578, "sampling/importance_sampling_ratio/min": 0.005101828370243311, "sampling/importance_sampling_ratio/mean": 1.021235466003418, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9896829351782799, "clip_ratio/low_mean": 0.02757352963089943, "clip_ratio/low_min": 0.02757352963089943, "clip_ratio/high_mean": 0.1027968917042017, "clip_ratio/high_max": 0.1027968917042017, "clip_ratio/region_mean": 0.13037042133510113, "reward_total_mean": 0.5409913063049316, "reward_meter_mean": 0.9619123935699463, "reward_meter_std": 0.02976352907717228, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9517722725868225, "reward_repeat_soft_std": 0.04128618910908699, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.5409913063049316, "reward_total_composite_std": 0.231267049908638} {"timestamp_utc": "2026-04-13T09:53:40Z", "mode": "train", "global_step": 1014, "epoch": 0.1018583626318433, "loss": -0.001, "grad_norm": 7.80047607421875, "learning_rate": 6.930303030303031e-06, "num_tokens": 1791389.0, "completions/mean_length": 104.375, "completions/min_length": 93.0, "completions/max_length": 117.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 104.375, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.900068998336792, "rewards/meter/std": 0.18561285734176636, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9651224613189697, "rewards/repeat_soft/std": 0.020892761647701263, "rewards/judge_quality/mean": 0.6074999570846558, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.613532543182373, "rewards/total_composite/std": 0.2619045674800873, "reward": 0.613532543182373, "reward_std": 0.2619045674800873, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13769012689590454, "sampling/sampling_logp_difference/max": 1.6550350189208984, "sampling/importance_sampling_ratio/min": 0.1910853534936905, "sampling/importance_sampling_ratio/mean": 1.0003539323806763, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9395716786384583, "clip_ratio/low_mean": 0.04137605335563421, "clip_ratio/low_min": 0.04137605335563421, "clip_ratio/high_mean": 0.09778095968067646, "clip_ratio/high_max": 0.09778095968067646, "clip_ratio/region_mean": 0.13915701303631067, "reward_total_mean": 0.613532543182373, "reward_meter_mean": 0.900068998336792, "reward_meter_std": 0.18561285734176636, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9651224613189697, "reward_repeat_soft_std": 0.020892761647701263, "reward_judge_quality_mean": 0.6074999570846558, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.613532543182373, "reward_total_composite_std": 0.2619045674800873} {"timestamp_utc": "2026-04-13T09:53:47Z", "mode": "train", "global_step": 1015, "epoch": 0.10195881466599699, "loss": -0.0082, "grad_norm": 8.329638481140137, "learning_rate": 6.927272727272728e-06, "num_tokens": 1793857.0, "completions/mean_length": 113.5, "completions/min_length": 102.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 113.5, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.9305871725082397, "rewards/meter/std": 0.13200393319129944, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9458436369895935, "rewards/repeat_soft/std": 0.05212901160120964, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5292602181434631, "rewards/total_composite/std": 0.03420192375779152, "reward": 0.5292602181434631, "reward_std": 0.03420192748308182, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14664679765701294, "sampling/sampling_logp_difference/max": 1.6847548484802246, "sampling/importance_sampling_ratio/min": 0.18548989295959473, "sampling/importance_sampling_ratio/mean": 1.0259116888046265, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9773837104439735, "clip_ratio/low_mean": 0.03706501983106136, "clip_ratio/low_min": 0.03706501983106136, "clip_ratio/high_mean": 0.08734916616231203, "clip_ratio/high_max": 0.08734916616231203, "clip_ratio/region_mean": 0.1244141859933734, "reward_total_mean": 0.5292602181434631, "reward_meter_mean": 0.9305871725082397, "reward_meter_std": 0.13200393319129944, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9458436369895935, "reward_repeat_soft_std": 0.05212901160120964, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5292602181434631, "reward_total_composite_std": 0.03420192375779152} {"timestamp_utc": "2026-04-13T09:53:54Z", "mode": "train", "global_step": 1016, "epoch": 0.10205926670015068, "loss": 0.0519, "grad_norm": 17.249526977539062, "learning_rate": 6.9242424242424245e-06, "num_tokens": 1795389.0, "completions/mean_length": 40.5, "completions/min_length": 36.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.5, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.46202582120895386, "rewards/meter/std": 0.35253575444221497, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9827362895011902, "rewards/repeat_soft/std": 0.013210390694439411, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.48027363419532776, "rewards/total_composite/std": 0.09445100277662277, "reward": 0.48027363419532776, "reward_std": 0.09445099532604218, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1720263659954071, "sampling/sampling_logp_difference/max": 1.777175784111023, "sampling/importance_sampling_ratio/min": 0.1691150963306427, "sampling/importance_sampling_ratio/mean": 1.011866807937622, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9946768060326576, "clip_ratio/low_mean": 0.12039735168218613, "clip_ratio/low_min": 0.12039735168218613, "clip_ratio/high_mean": 0.08580685593187809, "clip_ratio/high_max": 0.08580685593187809, "clip_ratio/region_mean": 0.20620420761406422, "reward_total_mean": 0.48027363419532776, "reward_meter_mean": 0.46202582120895386, "reward_meter_std": 0.35253575444221497, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9827362895011902, "reward_repeat_soft_std": 0.013210390694439411, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.48027363419532776, "reward_total_composite_std": 0.09445100277662277} {"timestamp_utc": "2026-04-13T09:54:00Z", "mode": "train", "global_step": 1017, "epoch": 0.10215971873430436, "loss": 0.0164, "grad_norm": 17.912460327148438, "learning_rate": 6.921212121212122e-06, "num_tokens": 1796921.0, "completions/mean_length": 36.5, "completions/min_length": 30.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.5, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.8674060702323914, "rewards/meter/std": 0.16231243312358856, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9547463059425354, "rewards/repeat_soft/std": 0.047527726739645004, "rewards/judge_quality/mean": 0.706250011920929, "rewards/judge_quality/std": 0.19316445291042328, "rewards/total_composite/mean": 0.7387485504150391, "rewards/total_composite/std": 0.12863102555274963, "reward": 0.7387485504150391, "reward_std": 0.12863105535507202, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18040382862091064, "sampling/sampling_logp_difference/max": 1.5863866806030273, "sampling/importance_sampling_ratio/min": 0.20466379821300507, "sampling/importance_sampling_ratio/mean": 1.0324939489364624, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2933059111237526, "clip_ratio/low_mean": 0.07632113993167877, "clip_ratio/low_min": 0.07632113993167877, "clip_ratio/high_mean": 0.12189750000834465, "clip_ratio/high_max": 0.12189750000834465, "clip_ratio/region_mean": 0.19821863994002342, "reward_total_mean": 0.7387485504150391, "reward_meter_mean": 0.8674060702323914, "reward_meter_std": 0.16231243312358856, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9547463059425354, "reward_repeat_soft_std": 0.047527726739645004, "reward_judge_quality_mean": 0.706250011920929, "reward_judge_quality_std": 0.19316445291042328, "reward_total_composite_mean": 0.7387485504150391, "reward_total_composite_std": 0.12863102555274963} {"timestamp_utc": "2026-04-13T09:54:07Z", "mode": "train", "global_step": 1018, "epoch": 0.10226017076845806, "loss": 0.0239, "grad_norm": 9.51146125793457, "learning_rate": 6.918181818181818e-06, "num_tokens": 1798948.0, "completions/mean_length": 82.375, "completions/min_length": 70.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.375, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.8627983331680298, "rewards/meter/std": 0.27659159898757935, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9733031988143921, "rewards/repeat_soft/std": 0.023586789146065712, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5315394401550293, "rewards/total_composite/std": 0.07479587197303772, "reward": 0.5315394401550293, "reward_std": 0.07479587942361832, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1614338755607605, "sampling/sampling_logp_difference/max": 2.192643165588379, "sampling/importance_sampling_ratio/min": 0.11162132769823074, "sampling/importance_sampling_ratio/mean": 1.003227710723877, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2144625186920166, "clip_ratio/low_mean": 0.028445051982998848, "clip_ratio/low_min": 0.028445051982998848, "clip_ratio/high_mean": 0.15170305408537388, "clip_ratio/high_max": 0.15170305408537388, "clip_ratio/region_mean": 0.18014810606837273, "reward_total_mean": 0.5315394401550293, "reward_meter_mean": 0.8627983331680298, "reward_meter_std": 0.27659159898757935, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9733031988143921, "reward_repeat_soft_std": 0.023586789146065712, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5315394401550293, "reward_total_composite_std": 0.07479587197303772} {"timestamp_utc": "2026-04-13T09:54:14Z", "mode": "train", "global_step": 1019, "epoch": 0.10236062280261175, "loss": 0.013, "grad_norm": 10.348309516906738, "learning_rate": 6.915151515151515e-06, "num_tokens": 1801016.0, "completions/mean_length": 91.5, "completions/min_length": 69.0, "completions/max_length": 120.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.5, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 120.0, "rewards/meter/mean": 0.6451882123947144, "rewards/meter/std": 0.2994335889816284, "rewards/count_adherence/mean": 0.84375, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8715066909790039, "rewards/repeat_soft/std": 0.10998848080635071, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.47752851247787476, "rewards/total_composite/std": 0.08254332095384598, "reward": 0.47752851247787476, "reward_std": 0.08254332095384598, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1572115123271942, "sampling/sampling_logp_difference/max": 2.785656213760376, "sampling/importance_sampling_ratio/min": 0.06168859452009201, "sampling/importance_sampling_ratio/mean": 1.021525502204895, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1143873408436775, "clip_ratio/low_mean": 0.05920429341495037, "clip_ratio/low_min": 0.05920429341495037, "clip_ratio/high_mean": 0.057869311422109604, "clip_ratio/high_max": 0.057869311422109604, "clip_ratio/region_mean": 0.11707360483705997, "reward_total_mean": 0.47752851247787476, "reward_meter_mean": 0.6451882123947144, "reward_meter_std": 0.2994335889816284, "reward_count_adherence_mean": 0.84375, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8715066909790039, "reward_repeat_soft_std": 0.10998848080635071, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.47752851247787476, "reward_total_composite_std": 0.08254332095384598} {"timestamp_utc": "2026-04-13T09:54:20Z", "mode": "train", "global_step": 1020, "epoch": 0.10246107483676545, "loss": 0.0311, "grad_norm": 8.36316204071045, "learning_rate": 6.912121212121212e-06, "num_tokens": 1803200.0, "completions/mean_length": 94.0, "completions/min_length": 84.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.0, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.769997239112854, "rewards/meter/std": 0.2617892026901245, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9365980625152588, "rewards/repeat_soft/std": 0.026574136689305305, "rewards/judge_quality/mean": 0.5699999928474426, "rewards/judge_quality/std": 0.16035676002502441, "rewards/total_composite/mean": 0.546271026134491, "rewards/total_composite/std": 0.1045205146074295, "reward": 0.546271026134491, "reward_std": 0.1045205146074295, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14200660586357117, "sampling/sampling_logp_difference/max": 1.6306532621383667, "sampling/importance_sampling_ratio/min": 0.19580160081386566, "sampling/importance_sampling_ratio/mean": 1.0257067680358887, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.14267548173666, "clip_ratio/low_mean": 0.07997111836448312, "clip_ratio/low_min": 0.07997111836448312, "clip_ratio/high_mean": 0.029277191497385502, "clip_ratio/high_max": 0.029277191497385502, "clip_ratio/region_mean": 0.10924830986186862, "reward_total_mean": 0.546271026134491, "reward_meter_mean": 0.769997239112854, "reward_meter_std": 0.2617892026901245, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9365980625152588, "reward_repeat_soft_std": 0.026574136689305305, "reward_judge_quality_mean": 0.5699999928474426, "reward_judge_quality_std": 0.16035676002502441, "reward_total_composite_mean": 0.546271026134491, "reward_total_composite_std": 0.1045205146074295} {"timestamp_utc": "2026-04-13T09:54:27Z", "mode": "train", "global_step": 1021, "epoch": 0.10256152687091914, "loss": 0.0667, "grad_norm": 11.773134231567383, "learning_rate": 6.90909090909091e-06, "num_tokens": 1805389.0, "completions/mean_length": 81.625, "completions/min_length": 48.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 81.625, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.7907544374465942, "rewards/meter/std": 0.28326550126075745, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9537135362625122, "rewards/repeat_soft/std": 0.015823470428586006, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.12906256318092346, "rewards/total_composite/mean": 0.5364459156990051, "rewards/total_composite/std": 0.06541776657104492, "reward": 0.5364459156990051, "reward_std": 0.06541776657104492, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15752579271793365, "sampling/sampling_logp_difference/max": 1.6947920322418213, "sampling/importance_sampling_ratio/min": 0.18363742530345917, "sampling/importance_sampling_ratio/mean": 1.0146045684814453, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0685177594423294, "clip_ratio/low_mean": 0.04912734962999821, "clip_ratio/low_min": 0.04912734962999821, "clip_ratio/high_mean": 0.10065225791186094, "clip_ratio/high_max": 0.10065225791186094, "clip_ratio/region_mean": 0.14977960754185915, "reward_total_mean": 0.5364459156990051, "reward_meter_mean": 0.7907544374465942, "reward_meter_std": 0.28326550126075745, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9537135362625122, "reward_repeat_soft_std": 0.015823470428586006, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.12906256318092346, "reward_total_composite_mean": 0.5364459156990051, "reward_total_composite_std": 0.06541776657104492} {"timestamp_utc": "2026-04-13T09:54:33Z", "mode": "train", "global_step": 1022, "epoch": 0.10266197890507282, "loss": 0.0067, "grad_norm": 15.89610481262207, "learning_rate": 6.906060606060606e-06, "num_tokens": 1806966.0, "completions/mean_length": 47.125, "completions/min_length": 39.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.125, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.5483500957489014, "rewards/meter/std": 0.42771151661872864, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9534322619438171, "rewards/repeat_soft/std": 0.053940366953611374, "rewards/judge_quality/mean": 0.4762499928474426, "rewards/judge_quality/std": 0.09941796213388443, "rewards/total_composite/mean": 0.5007228851318359, "rewards/total_composite/std": 0.11496507376432419, "reward": 0.5007228851318359, "reward_std": 0.11496507376432419, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16265255212783813, "sampling/sampling_logp_difference/max": 1.4988012313842773, "sampling/importance_sampling_ratio/min": 0.22339780628681183, "sampling/importance_sampling_ratio/mean": 1.0458287000656128, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0818421095609665, "clip_ratio/low_mean": 0.08918740693479776, "clip_ratio/low_min": 0.08918740693479776, "clip_ratio/high_mean": 0.06208793632686138, "clip_ratio/high_max": 0.06208793632686138, "clip_ratio/region_mean": 0.15127534326165915, "reward_total_mean": 0.5007228851318359, "reward_meter_mean": 0.5483500957489014, "reward_meter_std": 0.42771151661872864, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9534322619438171, "reward_repeat_soft_std": 0.053940366953611374, "reward_judge_quality_mean": 0.4762499928474426, "reward_judge_quality_std": 0.09941796213388443, "reward_total_composite_mean": 0.5007228851318359, "reward_total_composite_std": 0.11496507376432419} {"timestamp_utc": "2026-04-13T09:54:39Z", "mode": "train", "global_step": 1023, "epoch": 0.10276243093922652, "loss": 0.0205, "grad_norm": 14.367979049682617, "learning_rate": 6.903030303030304e-06, "num_tokens": 1808554.0, "completions/mean_length": 39.5, "completions/min_length": 32.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.5, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.8621435165405273, "rewards/meter/std": 0.30848589539527893, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9868086576461792, "rewards/repeat_soft/std": 0.010005839169025421, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.10392305999994278, "rewards/total_composite/mean": 0.6074839234352112, "rewards/total_composite/std": 0.10497891902923584, "reward": 0.6074839234352112, "reward_std": 0.10497891902923584, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16451585292816162, "sampling/sampling_logp_difference/max": 2.0143141746520996, "sampling/importance_sampling_ratio/min": 0.13341186940670013, "sampling/importance_sampling_ratio/mean": 1.035903811454773, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4066985547542572, "clip_ratio/low_mean": 0.008522727526724339, "clip_ratio/low_min": 0.008522727526724339, "clip_ratio/high_mean": 0.12929297517985106, "clip_ratio/high_max": 0.12929297517985106, "clip_ratio/region_mean": 0.1378157027065754, "reward_total_mean": 0.6074839234352112, "reward_meter_mean": 0.8621435165405273, "reward_meter_std": 0.30848589539527893, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9868086576461792, "reward_repeat_soft_std": 0.010005839169025421, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.10392305999994278, "reward_total_composite_mean": 0.6074839234352112, "reward_total_composite_std": 0.10497891902923584} {"timestamp_utc": "2026-04-13T09:54:44Z", "mode": "train", "global_step": 1024, "epoch": 0.10286288297338021, "loss": -0.0067, "grad_norm": 17.783018112182617, "learning_rate": 6.9e-06, "num_tokens": 1809928.0, "completions/mean_length": 25.75, "completions/min_length": 22.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.75, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9660782814025879, "rewards/meter/std": 0.06317251920700073, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9620115756988525, "rewards/repeat_soft/std": 0.0013815105194225907, "rewards/judge_quality/mean": 0.5637500286102295, "rewards/judge_quality/std": 0.22012579441070557, "rewards/total_composite/mean": 0.7001445293426514, "rewards/total_composite/std": 0.1472204178571701, "reward": 0.7001445293426514, "reward_std": 0.1472204029560089, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14203289151191711, "sampling/sampling_logp_difference/max": 1.0836036205291748, "sampling/importance_sampling_ratio/min": 0.35814791917800903, "sampling/importance_sampling_ratio/mean": 1.026048183441162, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9504631012678146, "clip_ratio/low_mean": 0.08692905679345131, "clip_ratio/low_min": 0.08692905679345131, "clip_ratio/high_mean": 0.017241379246115685, "clip_ratio/high_max": 0.017241379246115685, "clip_ratio/region_mean": 0.104170436039567, "reward_total_mean": 0.7001445293426514, "reward_meter_mean": 0.9660782814025879, "reward_meter_std": 0.06317251920700073, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9620115756988525, "reward_repeat_soft_std": 0.0013815105194225907, "reward_judge_quality_mean": 0.5637500286102295, "reward_judge_quality_std": 0.22012579441070557, "reward_total_composite_mean": 0.7001445293426514, "reward_total_composite_std": 0.1472204178571701} {"timestamp_utc": "2026-04-13T09:54:50Z", "mode": "train", "global_step": 1025, "epoch": 0.10296333500753391, "loss": 0.0915, "grad_norm": 16.288047790527344, "learning_rate": 6.896969696969697e-06, "num_tokens": 1811509.0, "completions/mean_length": 45.625, "completions/min_length": 36.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.625, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.736138105392456, "rewards/meter/std": 0.43030664324760437, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9859573841094971, "rewards/repeat_soft/std": 0.009918374940752983, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.5562318563461304, "rewards/total_composite/std": 0.11651643365621567, "reward": 0.5562318563461304, "reward_std": 0.11651644110679626, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15458832681179047, "sampling/sampling_logp_difference/max": 1.1145048141479492, "sampling/importance_sampling_ratio/min": 0.3280777037143707, "sampling/importance_sampling_ratio/mean": 1.0327446460723877, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4500950127840042, "clip_ratio/low_mean": 0.05376512371003628, "clip_ratio/low_min": 0.05376512371003628, "clip_ratio/high_mean": 0.12403824273496866, "clip_ratio/high_max": 0.12403824273496866, "clip_ratio/region_mean": 0.17780336644500494, "reward_total_mean": 0.5562318563461304, "reward_meter_mean": 0.736138105392456, "reward_meter_std": 0.43030664324760437, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9859573841094971, "reward_repeat_soft_std": 0.009918374940752983, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.5562318563461304, "reward_total_composite_std": 0.11651643365621567} {"timestamp_utc": "2026-04-13T09:54:56Z", "mode": "train", "global_step": 1026, "epoch": 0.10306378704168759, "loss": 0.2184, "grad_norm": 21.804195404052734, "learning_rate": 6.893939393939395e-06, "num_tokens": 1812732.0, "completions/mean_length": 26.875, "completions/min_length": 18.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.875, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.9173821210861206, "rewards/meter/std": 0.07125087082386017, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9528553485870361, "rewards/repeat_soft/std": 0.030632054433226585, "rewards/judge_quality/mean": 0.4099999964237213, "rewards/judge_quality/std": 0.06633248925209045, "rewards/total_composite/mean": 0.5611845254898071, "rewards/total_composite/std": 0.08697989583015442, "reward": 0.5611845254898071, "reward_std": 0.08697989583015442, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18380378186702728, "sampling/sampling_logp_difference/max": 0.9918375015258789, "sampling/importance_sampling_ratio/min": 0.37089455127716064, "sampling/importance_sampling_ratio/mean": 1.047836422920227, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3717611581087112, "clip_ratio/low_mean": 0.05922025255858898, "clip_ratio/low_min": 0.05922025255858898, "clip_ratio/high_mean": 0.12956710113212466, "clip_ratio/high_max": 0.12956710113212466, "clip_ratio/region_mean": 0.18878735369071364, "reward_total_mean": 0.5611845254898071, "reward_meter_mean": 0.9173821210861206, "reward_meter_std": 0.07125087082386017, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9528553485870361, "reward_repeat_soft_std": 0.030632054433226585, "reward_judge_quality_mean": 0.4099999964237213, "reward_judge_quality_std": 0.06633248925209045, "reward_total_composite_mean": 0.5611845254898071, "reward_total_composite_std": 0.08697989583015442} {"timestamp_utc": "2026-04-13T09:55:03Z", "mode": "train", "global_step": 1027, "epoch": 0.10316423907584128, "loss": 0.0335, "grad_norm": 14.44566535949707, "learning_rate": 6.890909090909092e-06, "num_tokens": 1814942.0, "completions/mean_length": 87.25, "completions/min_length": 79.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.25, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.6665823459625244, "rewards/meter/std": 0.319399356842041, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9894012808799744, "rewards/repeat_soft/std": 0.005543151404708624, "rewards/judge_quality/mean": 0.6525000333786011, "rewards/judge_quality/std": 0.20672619342803955, "rewards/total_composite/mean": 0.5615822672843933, "rewards/total_composite/std": 0.1492321789264679, "reward": 0.5615822672843933, "reward_std": 0.1492321789264679, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18131457269191742, "sampling/sampling_logp_difference/max": 2.362147331237793, "sampling/importance_sampling_ratio/min": 0.0942176878452301, "sampling/importance_sampling_ratio/mean": 1.0077650547027588, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8842276111245155, "clip_ratio/low_mean": 0.10200242698192596, "clip_ratio/low_min": 0.10200242698192596, "clip_ratio/high_mean": 0.050042061135172844, "clip_ratio/high_max": 0.050042061135172844, "clip_ratio/region_mean": 0.1520444881170988, "reward_total_mean": 0.5615822672843933, "reward_meter_mean": 0.6665823459625244, "reward_meter_std": 0.319399356842041, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9894012808799744, "reward_repeat_soft_std": 0.005543151404708624, "reward_judge_quality_mean": 0.6525000333786011, "reward_judge_quality_std": 0.20672619342803955, "reward_total_composite_mean": 0.5615822672843933, "reward_total_composite_std": 0.1492321789264679} {"timestamp_utc": "2026-04-13T09:55:09Z", "mode": "train", "global_step": 1028, "epoch": 0.10326469110999498, "loss": 0.0588, "grad_norm": 8.865718841552734, "learning_rate": 6.887878787878789e-06, "num_tokens": 1817006.0, "completions/mean_length": 75.0, "completions/min_length": 68.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.0, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.9105406999588013, "rewards/meter/std": 0.11760742217302322, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9534875154495239, "rewards/repeat_soft/std": 0.04239608719944954, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.6329624652862549, "rewards/total_composite/std": 0.07994944602251053, "reward": 0.6329624652862549, "reward_std": 0.07994943857192993, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15391583740711212, "sampling/sampling_logp_difference/max": 1.6568903923034668, "sampling/importance_sampling_ratio/min": 0.19073115289211273, "sampling/importance_sampling_ratio/mean": 1.0367811918258667, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.289549097418785, "clip_ratio/low_mean": 0.09470611996948719, "clip_ratio/low_min": 0.09470611996948719, "clip_ratio/high_mean": 0.043478261679410934, "clip_ratio/high_max": 0.043478261679410934, "clip_ratio/region_mean": 0.13818438164889812, "reward_total_mean": 0.6329624652862549, "reward_meter_mean": 0.9105406999588013, "reward_meter_std": 0.11760742217302322, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9534875154495239, "reward_repeat_soft_std": 0.04239608719944954, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.6329624652862549, "reward_total_composite_std": 0.07994944602251053} {"timestamp_utc": "2026-04-13T09:55:16Z", "mode": "train", "global_step": 1029, "epoch": 0.10336514314414867, "loss": 0.0274, "grad_norm": 9.817951202392578, "learning_rate": 6.8848484848484854e-06, "num_tokens": 1818995.0, "completions/mean_length": 73.625, "completions/min_length": 52.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.625, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.9692832231521606, "rewards/meter/std": 0.03116937167942524, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9359431266784668, "rewards/repeat_soft/std": 0.04442360997200012, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.12906256318092346, "rewards/total_composite/mean": 0.5677253007888794, "rewards/total_composite/std": 0.08369077742099762, "reward": 0.5677253007888794, "reward_std": 0.08369076997041702, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15762563049793243, "sampling/sampling_logp_difference/max": 1.8013087511062622, "sampling/importance_sampling_ratio/min": 0.16508269309997559, "sampling/importance_sampling_ratio/mean": 1.0174403190612793, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.230010949075222, "clip_ratio/low_mean": 0.11425292678177357, "clip_ratio/low_min": 0.11425292678177357, "clip_ratio/high_mean": 0.03355263266712427, "clip_ratio/high_max": 0.03355263266712427, "clip_ratio/region_mean": 0.14780555944889784, "reward_total_mean": 0.5677253007888794, "reward_meter_mean": 0.9692832231521606, "reward_meter_std": 0.03116937167942524, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9359431266784668, "reward_repeat_soft_std": 0.04442360997200012, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.12906256318092346, "reward_total_composite_mean": 0.5677253007888794, "reward_total_composite_std": 0.08369077742099762} {"timestamp_utc": "2026-04-13T09:55:22Z", "mode": "train", "global_step": 1030, "epoch": 0.10346559517830237, "loss": 0.109, "grad_norm": 14.510078430175781, "learning_rate": 6.881818181818183e-06, "num_tokens": 1820599.0, "completions/mean_length": 42.5, "completions/min_length": 30.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.5, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.7224732637405396, "rewards/meter/std": 0.3413645029067993, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9543820023536682, "rewards/repeat_soft/std": 0.04431818425655365, "rewards/judge_quality/mean": 0.6024999618530273, "rewards/judge_quality/std": 0.1976107507944107, "rewards/total_composite/mean": 0.6407123804092407, "rewards/total_composite/std": 0.18194904923439026, "reward": 0.6407123804092407, "reward_std": 0.18194904923439026, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1599501073360443, "sampling/sampling_logp_difference/max": 1.005096435546875, "sampling/importance_sampling_ratio/min": 0.36600935459136963, "sampling/importance_sampling_ratio/mean": 1.0132290124893188, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3126289695501328, "clip_ratio/low_mean": 0.08708263840526342, "clip_ratio/low_min": 0.08708263840526342, "clip_ratio/high_mean": 0.05918329767882824, "clip_ratio/high_max": 0.05918329767882824, "clip_ratio/region_mean": 0.14626593608409166, "reward_total_mean": 0.6407123804092407, "reward_meter_mean": 0.7224732637405396, "reward_meter_std": 0.3413645029067993, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9543820023536682, "reward_repeat_soft_std": 0.04431818425655365, "reward_judge_quality_mean": 0.6024999618530273, "reward_judge_quality_std": 0.1976107507944107, "reward_total_composite_mean": 0.6407123804092407, "reward_total_composite_std": 0.18194904923439026} {"timestamp_utc": "2026-04-13T09:55:28Z", "mode": "train", "global_step": 1031, "epoch": 0.10356604721245605, "loss": -0.0032, "grad_norm": 13.197099685668945, "learning_rate": 6.878787878787879e-06, "num_tokens": 1822347.0, "completions/mean_length": 62.5, "completions/min_length": 55.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.5, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9130160808563232, "rewards/meter/std": 0.21048715710639954, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.930400013923645, "rewards/repeat_soft/std": 0.02374812588095665, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5388133525848389, "rewards/total_composite/std": 0.05753803625702858, "reward": 0.5388133525848389, "reward_std": 0.05753804370760918, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15961933135986328, "sampling/sampling_logp_difference/max": 1.3118410110473633, "sampling/importance_sampling_ratio/min": 0.26932376623153687, "sampling/importance_sampling_ratio/mean": 1.0412040948867798, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4632093608379364, "clip_ratio/low_mean": 0.015909090638160706, "clip_ratio/low_min": 0.015909090638160706, "clip_ratio/high_mean": 0.11567418463528156, "clip_ratio/high_max": 0.11567418463528156, "clip_ratio/region_mean": 0.13158327527344227, "reward_total_mean": 0.5388133525848389, "reward_meter_mean": 0.9130160808563232, "reward_meter_std": 0.21048715710639954, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.930400013923645, "reward_repeat_soft_std": 0.02374812588095665, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5388133525848389, "reward_total_composite_std": 0.05753803625702858} {"timestamp_utc": "2026-04-13T09:55:39Z", "mode": "train", "global_step": 1032, "epoch": 0.10366649924660974, "loss": -0.1328, "grad_norm": 2.621129035949707, "learning_rate": 6.875757575757576e-06, "num_tokens": 1824186.0, "completions/mean_length": 117.875, "completions/min_length": 54.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 61.57143020629883, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.8732489347457886, "rewards/meter/std": 0.3334799110889435, "rewards/count_adherence/mean": 0.71875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9191628694534302, "rewards/repeat_soft/std": 0.04863610491156578, "rewards/judge_quality/mean": 0.3687499761581421, "rewards/judge_quality/std": 0.1940867006778717, "rewards/total_composite/mean": 0.4840266704559326, "rewards/total_composite/std": 0.21617349982261658, "reward": 0.4840266704559326, "reward_std": 0.21617348492145538, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15878494083881378, "sampling/sampling_logp_difference/max": 2.8949594497680664, "sampling/importance_sampling_ratio/min": 0.05530126392841339, "sampling/importance_sampling_ratio/mean": 1.0299222469329834, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0180036127567291, "clip_ratio/low_mean": 0.024951783940196037, "clip_ratio/low_min": 0.024951783940196037, "clip_ratio/high_mean": 0.08247385453432798, "clip_ratio/high_max": 0.08247385453432798, "clip_ratio/region_mean": 0.10742563847452402, "reward_total_mean": 0.4840266704559326, "reward_meter_mean": 0.8732489347457886, "reward_meter_std": 0.3334799110889435, "reward_count_adherence_mean": 0.71875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9191628694534302, "reward_repeat_soft_std": 0.04863610491156578, "reward_judge_quality_mean": 0.3687499761581421, "reward_judge_quality_std": 0.1940867006778717, "reward_total_composite_mean": 0.4840266704559326, "reward_total_composite_std": 0.21617349982261658} {"timestamp_utc": "2026-04-13T09:55:50Z", "mode": "train", "global_step": 1033, "epoch": 0.10376695128076344, "loss": -0.0549, "grad_norm": 4.162497043609619, "learning_rate": 6.872727272727273e-06, "num_tokens": 1825457.0, "completions/mean_length": 84.875, "completions/min_length": 21.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 23.85714340209961, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.37773576378822327, "rewards/meter/std": 0.4795103371143341, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9588949680328369, "rewards/repeat_soft/std": 0.024640807881951332, "rewards/judge_quality/mean": 0.38875001668930054, "rewards/judge_quality/std": 0.262593537569046, "rewards/total_composite/mean": 0.4405129849910736, "rewards/total_composite/std": 0.26010143756866455, "reward": 0.4405129849910736, "reward_std": 0.26010146737098694, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1446915566921234, "sampling/sampling_logp_difference/max": 0.833672285079956, "sampling/importance_sampling_ratio/min": 0.47228142619132996, "sampling/importance_sampling_ratio/mean": 1.040169358253479, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.445899561047554, "clip_ratio/low_mean": 0.07106978353112936, "clip_ratio/low_min": 0.07106978353112936, "clip_ratio/high_mean": 0.06505952402949333, "clip_ratio/high_max": 0.06505952402949333, "clip_ratio/region_mean": 0.1361293075606227, "reward_total_mean": 0.4405129849910736, "reward_meter_mean": 0.37773576378822327, "reward_meter_std": 0.4795103371143341, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9588949680328369, "reward_repeat_soft_std": 0.024640807881951332, "reward_judge_quality_mean": 0.38875001668930054, "reward_judge_quality_std": 0.262593537569046, "reward_total_composite_mean": 0.4405129849910736, "reward_total_composite_std": 0.26010143756866455} {"timestamp_utc": "2026-04-13T09:55:57Z", "mode": "train", "global_step": 1034, "epoch": 0.10386740331491713, "loss": 0.0458, "grad_norm": 12.486868858337402, "learning_rate": 6.869696969696971e-06, "num_tokens": 1827302.0, "completions/mean_length": 49.625, "completions/min_length": 38.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.625, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.9833406209945679, "rewards/meter/std": 0.02095111645758152, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9626805186271667, "rewards/repeat_soft/std": 0.032711006700992584, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.6223706007003784, "rewards/total_composite/std": 0.013778239488601685, "reward": 0.6223706007003784, "reward_std": 0.01377825252711773, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1560501903295517, "sampling/sampling_logp_difference/max": 1.6480932235717773, "sampling/importance_sampling_ratio/min": 0.19241644442081451, "sampling/importance_sampling_ratio/mean": 1.0278218984603882, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6125024110078812, "clip_ratio/low_mean": 0.09322284162044525, "clip_ratio/low_min": 0.09322284162044525, "clip_ratio/high_mean": 0.05953734088689089, "clip_ratio/high_max": 0.05953734088689089, "clip_ratio/region_mean": 0.15276018250733614, "reward_total_mean": 0.6223706007003784, "reward_meter_mean": 0.9833406209945679, "reward_meter_std": 0.02095111645758152, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9626805186271667, "reward_repeat_soft_std": 0.032711006700992584, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.6223706007003784, "reward_total_composite_std": 0.013778239488601685} {"timestamp_utc": "2026-04-13T09:56:03Z", "mode": "train", "global_step": 1035, "epoch": 0.10396785534907081, "loss": 0.0472, "grad_norm": 8.736976623535156, "learning_rate": 6.866666666666667e-06, "num_tokens": 1829630.0, "completions/mean_length": 96.0, "completions/min_length": 82.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.0, "completions/min_terminated_length": 82.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.8666692972183228, "rewards/meter/std": 0.3343394994735718, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8650614023208618, "rewards/repeat_soft/std": 0.12002670764923096, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.5006651878356934, "rewards/total_composite/std": 0.08232749998569489, "reward": 0.5006651878356934, "reward_std": 0.08232749998569489, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12479706853628159, "sampling/sampling_logp_difference/max": 1.4670276641845703, "sampling/importance_sampling_ratio/min": 0.2306099236011505, "sampling/importance_sampling_ratio/mean": 1.0250110626220703, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.084361806511879, "clip_ratio/low_mean": 0.019859813153743744, "clip_ratio/low_min": 0.019859813153743744, "clip_ratio/high_mean": 0.08743510372005403, "clip_ratio/high_max": 0.08743510372005403, "clip_ratio/region_mean": 0.10729491687379777, "reward_total_mean": 0.5006651878356934, "reward_meter_mean": 0.8666692972183228, "reward_meter_std": 0.3343394994735718, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8650614023208618, "reward_repeat_soft_std": 0.12002670764923096, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.5006651878356934, "reward_total_composite_std": 0.08232749998569489} {"timestamp_utc": "2026-04-13T09:56:10Z", "mode": "train", "global_step": 1036, "epoch": 0.1040683073832245, "loss": 0.0382, "grad_norm": 12.796359062194824, "learning_rate": 6.8636363636363645e-06, "num_tokens": 1831621.0, "completions/mean_length": 70.875, "completions/min_length": 57.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.875, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.9110418558120728, "rewards/meter/std": 0.198968306183815, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9122515916824341, "rewards/repeat_soft/std": 0.07234229147434235, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5355521440505981, "rewards/total_composite/std": 0.05388759821653366, "reward": 0.5355521440505981, "reward_std": 0.05388757586479187, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15520618855953217, "sampling/sampling_logp_difference/max": 1.6682729721069336, "sampling/importance_sampling_ratio/min": 0.1885724514722824, "sampling/importance_sampling_ratio/mean": 1.0142109394073486, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2992232292890549, "clip_ratio/low_mean": 0.021831881254911423, "clip_ratio/low_min": 0.021831881254911423, "clip_ratio/high_mean": 0.11172603955492377, "clip_ratio/high_max": 0.11172603955492377, "clip_ratio/region_mean": 0.1335579208098352, "reward_total_mean": 0.5355521440505981, "reward_meter_mean": 0.9110418558120728, "reward_meter_std": 0.198968306183815, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9122515916824341, "reward_repeat_soft_std": 0.07234229147434235, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5355521440505981, "reward_total_composite_std": 0.05388759821653366} {"timestamp_utc": "2026-04-13T09:56:16Z", "mode": "train", "global_step": 1037, "epoch": 0.1041687594173782, "loss": 0.0578, "grad_norm": 14.418715476989746, "learning_rate": 6.860606060606061e-06, "num_tokens": 1833571.0, "completions/mean_length": 62.75, "completions/min_length": 54.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.75, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.4608146548271179, "rewards/meter/std": 0.37401625514030457, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9413318037986755, "rewards/repeat_soft/std": 0.058841411024332047, "rewards/judge_quality/mean": 0.6237500309944153, "rewards/judge_quality/std": 0.2232191562652588, "rewards/total_composite/mean": 0.4488751292228699, "rewards/total_composite/std": 0.11704400181770325, "reward": 0.4488751292228699, "reward_std": 0.11704400926828384, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1682649403810501, "sampling/sampling_logp_difference/max": 2.2751240730285645, "sampling/importance_sampling_ratio/min": 0.1027841567993164, "sampling/importance_sampling_ratio/mean": 1.0111799240112305, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8135004937648773, "clip_ratio/low_mean": 0.0947592044249177, "clip_ratio/low_min": 0.0947592044249177, "clip_ratio/high_mean": 0.06098484992980957, "clip_ratio/high_max": 0.06098484992980957, "clip_ratio/region_mean": 0.15574405435472727, "reward_total_mean": 0.4488751292228699, "reward_meter_mean": 0.4608146548271179, "reward_meter_std": 0.37401625514030457, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9413318037986755, "reward_repeat_soft_std": 0.058841411024332047, "reward_judge_quality_mean": 0.6237500309944153, "reward_judge_quality_std": 0.2232191562652588, "reward_total_composite_mean": 0.4488751292228699, "reward_total_composite_std": 0.11704400181770325} {"timestamp_utc": "2026-04-13T09:56:29Z", "mode": "train", "global_step": 1038, "epoch": 0.1042692114515319, "loss": -0.1081, "grad_norm": 3.557382822036743, "learning_rate": 6.857575757575758e-06, "num_tokens": 1835343.0, "completions/mean_length": 101.5, "completions/min_length": 35.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 42.85714340209961, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.6340430974960327, "rewards/meter/std": 0.4002546966075897, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9766391515731812, "rewards/repeat_soft/std": 0.01620764099061489, "rewards/judge_quality/mean": 0.581250011920929, "rewards/judge_quality/std": 0.2953660190105438, "rewards/total_composite/mean": 0.5815249681472778, "rewards/total_composite/std": 0.2931891679763794, "reward": 0.5815249681472778, "reward_std": 0.293189138174057, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19768273830413818, "sampling/sampling_logp_difference/max": 1.267542839050293, "sampling/importance_sampling_ratio/min": 0.2815225422382355, "sampling/importance_sampling_ratio/mean": 1.0388178825378418, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5156055986881256, "clip_ratio/low_mean": 0.05267857201397419, "clip_ratio/low_min": 0.05267857201397419, "clip_ratio/high_mean": 0.08354822359979153, "clip_ratio/high_max": 0.08354822359979153, "clip_ratio/region_mean": 0.13622679561376572, "reward_total_mean": 0.5815249681472778, "reward_meter_mean": 0.6340430974960327, "reward_meter_std": 0.4002546966075897, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9766391515731812, "reward_repeat_soft_std": 0.01620764099061489, "reward_judge_quality_mean": 0.581250011920929, "reward_judge_quality_std": 0.2953660190105438, "reward_total_composite_mean": 0.5815249681472778, "reward_total_composite_std": 0.2931891679763794} {"timestamp_utc": "2026-04-13T09:56:35Z", "mode": "train", "global_step": 1039, "epoch": 0.10436966348568559, "loss": 0.0198, "grad_norm": 17.626461029052734, "learning_rate": 6.854545454545455e-06, "num_tokens": 1836971.0, "completions/mean_length": 34.5, "completions/min_length": 25.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.5, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.42783617973327637, "rewards/meter/std": 0.36010390520095825, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9915820360183716, "rewards/repeat_soft/std": 0.00942145474255085, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.17136012017726898, "rewards/total_composite/mean": 0.4947046637535095, "rewards/total_composite/std": 0.13748851418495178, "reward": 0.4947046637535095, "reward_std": 0.13748851418495178, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1619514524936676, "sampling/sampling_logp_difference/max": 1.6502609252929688, "sampling/importance_sampling_ratio/min": 0.19199980795383453, "sampling/importance_sampling_ratio/mean": 1.031578540802002, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2084343880414963, "clip_ratio/low_mean": 0.09601151198148727, "clip_ratio/low_min": 0.09601151198148727, "clip_ratio/high_mean": 0.06850469391793013, "clip_ratio/high_max": 0.06850469391793013, "clip_ratio/region_mean": 0.1645162058994174, "reward_total_mean": 0.4947046637535095, "reward_meter_mean": 0.42783617973327637, "reward_meter_std": 0.36010390520095825, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9915820360183716, "reward_repeat_soft_std": 0.00942145474255085, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.17136012017726898, "reward_total_composite_mean": 0.4947046637535095, "reward_total_composite_std": 0.13748851418495178} {"timestamp_utc": "2026-04-13T09:56:42Z", "mode": "train", "global_step": 1040, "epoch": 0.10447011551983927, "loss": 0.0609, "grad_norm": 10.139225006103516, "learning_rate": 6.851515151515153e-06, "num_tokens": 1839066.0, "completions/mean_length": 80.875, "completions/min_length": 74.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.875, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.7077836990356445, "rewards/meter/std": 0.39633890986442566, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9848485589027405, "rewards/repeat_soft/std": 0.015427964739501476, "rewards/judge_quality/mean": 0.5824999809265137, "rewards/judge_quality/std": 0.23260943591594696, "rewards/total_composite/mean": 0.5426092743873596, "rewards/total_composite/std": 0.16634903848171234, "reward": 0.5426092743873596, "reward_std": 0.16634905338287354, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17039991915225983, "sampling/sampling_logp_difference/max": 1.3873834609985352, "sampling/importance_sampling_ratio/min": 0.24972787499427795, "sampling/importance_sampling_ratio/mean": 1.0405789613723755, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9115580022335052, "clip_ratio/low_mean": 0.06445186212658882, "clip_ratio/low_min": 0.06445186212658882, "clip_ratio/high_mean": 0.09954247251152992, "clip_ratio/high_max": 0.09954247251152992, "clip_ratio/region_mean": 0.16399433463811874, "reward_total_mean": 0.5426092743873596, "reward_meter_mean": 0.7077836990356445, "reward_meter_std": 0.39633890986442566, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9848485589027405, "reward_repeat_soft_std": 0.015427964739501476, "reward_judge_quality_mean": 0.5824999809265137, "reward_judge_quality_std": 0.23260943591594696, "reward_total_composite_mean": 0.5426092743873596, "reward_total_composite_std": 0.16634903848171234} {"timestamp_utc": "2026-04-13T09:56:49Z", "mode": "train", "global_step": 1041, "epoch": 0.10457056755399297, "loss": -0.0033, "grad_norm": 8.641227722167969, "learning_rate": 6.848484848484849e-06, "num_tokens": 1841294.0, "completions/mean_length": 91.5, "completions/min_length": 78.0, "completions/max_length": 125.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.5, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.972457766532898, "rewards/meter/std": 0.04122280329465866, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9048149585723877, "rewards/repeat_soft/std": 0.02271733246743679, "rewards/judge_quality/mean": 0.38624998927116394, "rewards/judge_quality/std": 0.09545940905809402, "rewards/total_composite/mean": 0.5127357840538025, "rewards/total_composite/std": 0.06192810833454132, "reward": 0.5127357840538025, "reward_std": 0.06192811205983162, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17517006397247314, "sampling/sampling_logp_difference/max": 1.6087284088134766, "sampling/importance_sampling_ratio/min": 0.20014193654060364, "sampling/importance_sampling_ratio/mean": 1.0346416234970093, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7875735759735107, "clip_ratio/low_mean": 0.027623284608125687, "clip_ratio/low_min": 0.027623284608125687, "clip_ratio/high_mean": 0.1168333850800991, "clip_ratio/high_max": 0.1168333850800991, "clip_ratio/region_mean": 0.1444566696882248, "reward_total_mean": 0.5127357840538025, "reward_meter_mean": 0.972457766532898, "reward_meter_std": 0.04122280329465866, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9048149585723877, "reward_repeat_soft_std": 0.02271733246743679, "reward_judge_quality_mean": 0.38624998927116394, "reward_judge_quality_std": 0.09545940905809402, "reward_total_composite_mean": 0.5127357840538025, "reward_total_composite_std": 0.06192810833454132} {"timestamp_utc": "2026-04-13T09:56:55Z", "mode": "train", "global_step": 1042, "epoch": 0.10467101958814666, "loss": 0.0234, "grad_norm": 12.043434143066406, "learning_rate": 6.845454545454546e-06, "num_tokens": 1842849.0, "completions/mean_length": 43.375, "completions/min_length": 40.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.375, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.6926597356796265, "rewards/meter/std": 0.3405209481716156, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9692997932434082, "rewards/repeat_soft/std": 0.04518914222717285, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.6476796865463257, "rewards/total_composite/std": 0.226663738489151, "reward": 0.6476796865463257, "reward_std": 0.226663738489151, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11147625744342804, "sampling/sampling_logp_difference/max": 1.3472824096679688, "sampling/importance_sampling_ratio/min": 0.25994572043418884, "sampling/importance_sampling_ratio/mean": 1.0130397081375122, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7532584443688393, "clip_ratio/low_mean": 0.061964747961610556, "clip_ratio/low_min": 0.061964747961610556, "clip_ratio/high_mean": 0.025568853598088026, "clip_ratio/high_max": 0.025568853598088026, "clip_ratio/region_mean": 0.08753360155969858, "reward_total_mean": 0.6476796865463257, "reward_meter_mean": 0.6926597356796265, "reward_meter_std": 0.3405209481716156, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9692997932434082, "reward_repeat_soft_std": 0.04518914222717285, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.6476796865463257, "reward_total_composite_std": 0.226663738489151} {"timestamp_utc": "2026-04-13T09:57:02Z", "mode": "train", "global_step": 1043, "epoch": 0.10477147162230036, "loss": 0.0315, "grad_norm": 9.035298347473145, "learning_rate": 6.842424242424243e-06, "num_tokens": 1845063.0, "completions/mean_length": 87.75, "completions/min_length": 76.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.75, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.621731162071228, "rewards/meter/std": 0.1971936672925949, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.965823769569397, "rewards/repeat_soft/std": 0.02204710990190506, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.4653659462928772, "rewards/total_composite/std": 0.0575556755065918, "reward": 0.4653659462928772, "reward_std": 0.0575556717813015, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.172149658203125, "sampling/sampling_logp_difference/max": 1.8061906099319458, "sampling/importance_sampling_ratio/min": 0.16427874565124512, "sampling/importance_sampling_ratio/mean": 1.033522605895996, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5045871436595917, "clip_ratio/low_mean": 0.06947345472872257, "clip_ratio/low_min": 0.06947345472872257, "clip_ratio/high_mean": 0.0719126695767045, "clip_ratio/high_max": 0.0719126695767045, "clip_ratio/region_mean": 0.14138612430542707, "reward_total_mean": 0.4653659462928772, "reward_meter_mean": 0.621731162071228, "reward_meter_std": 0.1971936672925949, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.965823769569397, "reward_repeat_soft_std": 0.02204710990190506, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.4653659462928772, "reward_total_composite_std": 0.0575556755065918} {"timestamp_utc": "2026-04-13T09:57:08Z", "mode": "train", "global_step": 1044, "epoch": 0.10487192365645405, "loss": 0.0675, "grad_norm": 15.171388626098633, "learning_rate": 6.83939393939394e-06, "num_tokens": 1846734.0, "completions/mean_length": 38.875, "completions/min_length": 34.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.875, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.6489021182060242, "rewards/meter/std": 0.3967968821525574, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9885673522949219, "rewards/repeat_soft/std": 0.014239184558391571, "rewards/judge_quality/mean": 0.5175000429153442, "rewards/judge_quality/std": 0.1348809152841568, "rewards/total_composite/mean": 0.5683600902557373, "rewards/total_composite/std": 0.1472955346107483, "reward": 0.5683600902557373, "reward_std": 0.1472955346107483, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18410830199718475, "sampling/sampling_logp_difference/max": 1.4863176345825195, "sampling/importance_sampling_ratio/min": 0.22620409727096558, "sampling/importance_sampling_ratio/mean": 1.034193754196167, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.764886274933815, "clip_ratio/low_mean": 0.06886911019682884, "clip_ratio/low_min": 0.06886911019682884, "clip_ratio/high_mean": 0.08515406399965286, "clip_ratio/high_max": 0.08515406399965286, "clip_ratio/region_mean": 0.1540231741964817, "reward_total_mean": 0.5683600902557373, "reward_meter_mean": 0.6489021182060242, "reward_meter_std": 0.3967968821525574, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9885673522949219, "reward_repeat_soft_std": 0.014239184558391571, "reward_judge_quality_mean": 0.5175000429153442, "reward_judge_quality_std": 0.1348809152841568, "reward_total_composite_mean": 0.5683600902557373, "reward_total_composite_std": 0.1472955346107483} {"timestamp_utc": "2026-04-13T09:57:19Z", "mode": "train", "global_step": 1045, "epoch": 0.10497237569060773, "loss": -0.0614, "grad_norm": 1.2960211038589478, "learning_rate": 6.8363636363636364e-06, "num_tokens": 1848221.0, "completions/mean_length": 209.875, "completions/min_length": 26.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 28.600000381469727, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.7898814678192139, "rewards/meter/std": 0.33841297030448914, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.9671218991279602, "rewards/repeat_soft/std": 0.013286017812788486, "rewards/judge_quality/mean": 0.22500000894069672, "rewards/judge_quality/std": 0.16690459847450256, "rewards/total_composite/mean": 0.3426004648208618, "rewards/total_composite/std": 0.29025980830192566, "reward": 0.3426004648208618, "reward_std": 0.29025980830192566, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22148950397968292, "sampling/sampling_logp_difference/max": 1.2577190399169922, "sampling/importance_sampling_ratio/min": 0.2843017578125, "sampling/importance_sampling_ratio/mean": 1.036569595336914, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6734884232282639, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.11819592490792274, "clip_ratio/high_max": 0.11819592490792274, "clip_ratio/region_mean": 0.11819592490792274, "reward_total_mean": 0.3426004648208618, "reward_meter_mean": 0.7898814678192139, "reward_meter_std": 0.33841297030448914, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.9671218991279602, "reward_repeat_soft_std": 0.013286017812788486, "reward_judge_quality_mean": 0.22500000894069672, "reward_judge_quality_std": 0.16690459847450256, "reward_total_composite_mean": 0.3426004648208618, "reward_total_composite_std": 0.29025980830192566} {"timestamp_utc": "2026-04-13T09:57:30Z", "mode": "train", "global_step": 1046, "epoch": 0.10507282772476143, "loss": -0.1636, "grad_norm": 2.6663920879364014, "learning_rate": 6.833333333333334e-06, "num_tokens": 1849932.0, "completions/mean_length": 116.875, "completions/min_length": 54.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 60.42857360839844, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.8841900825500488, "rewards/meter/std": 0.2517623007297516, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.945469081401825, "rewards/repeat_soft/std": 0.06652350723743439, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.22947145998477936, "rewards/total_composite/mean": 0.5942699909210205, "rewards/total_composite/std": 0.2586207389831543, "reward": 0.5942699909210205, "reward_std": 0.2586207091808319, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1772335171699524, "sampling/sampling_logp_difference/max": 1.5601367950439453, "sampling/importance_sampling_ratio/min": 0.21010732650756836, "sampling/importance_sampling_ratio/mean": 1.0624712705612183, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7860264033079147, "clip_ratio/low_mean": 0.013636363670229912, "clip_ratio/low_min": 0.013636363670229912, "clip_ratio/high_mean": 0.1267325459048152, "clip_ratio/high_max": 0.1267325459048152, "clip_ratio/region_mean": 0.1403689095750451, "reward_total_mean": 0.5942699909210205, "reward_meter_mean": 0.8841900825500488, "reward_meter_std": 0.2517623007297516, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.945469081401825, "reward_repeat_soft_std": 0.06652350723743439, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.22947145998477936, "reward_total_composite_mean": 0.5942699909210205, "reward_total_composite_std": 0.2586207389831543} {"timestamp_utc": "2026-04-13T09:57:36Z", "mode": "train", "global_step": 1047, "epoch": 0.10517327975891512, "loss": 0.1043, "grad_norm": 21.364267349243164, "learning_rate": 6.83030303030303e-06, "num_tokens": 1851278.0, "completions/mean_length": 22.25, "completions/min_length": 18.0, "completions/max_length": 27.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.25, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.8054901957511902, "rewards/meter/std": 0.3369612991809845, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9038182497024536, "rewards/repeat_soft/std": 0.08919284492731094, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.17136012017726898, "rewards/total_composite/mean": 0.6021766662597656, "rewards/total_composite/std": 0.15866397321224213, "reward": 0.6021766662597656, "reward_std": 0.15866400301456451, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1650094836950302, "sampling/sampling_logp_difference/max": 1.4395227432250977, "sampling/importance_sampling_ratio/min": 0.23704086244106293, "sampling/importance_sampling_ratio/mean": 0.9973371028900146, "sampling/importance_sampling_ratio/max": 1.6196039915084839, "entropy": 1.2577237635850906, "clip_ratio/low_mean": 0.05333129828795791, "clip_ratio/low_min": 0.05333129828795791, "clip_ratio/high_mean": 0.07839912362396717, "clip_ratio/high_max": 0.07839912362396717, "clip_ratio/region_mean": 0.13173042191192508, "reward_total_mean": 0.6021766662597656, "reward_meter_mean": 0.8054901957511902, "reward_meter_std": 0.3369612991809845, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9038182497024536, "reward_repeat_soft_std": 0.08919284492731094, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.17136012017726898, "reward_total_composite_mean": 0.6021766662597656, "reward_total_composite_std": 0.15866397321224213} {"timestamp_utc": "2026-04-13T09:57:47Z", "mode": "train", "global_step": 1048, "epoch": 0.10527373179306881, "loss": -0.1363, "grad_norm": 3.268489360809326, "learning_rate": 6.827272727272728e-06, "num_tokens": 1852978.0, "completions/mean_length": 119.5, "completions/min_length": 53.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 63.42857360839844, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.7645368576049805, "rewards/meter/std": 0.29501092433929443, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9716801643371582, "rewards/repeat_soft/std": 0.014356398023664951, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.22947144508361816, "rewards/total_composite/mean": 0.5657984614372253, "rewards/total_composite/std": 0.2622303068637848, "reward": 0.5657984614372253, "reward_std": 0.2622303366661072, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1616942435503006, "sampling/sampling_logp_difference/max": 1.4891023635864258, "sampling/importance_sampling_ratio/min": 0.2255750596523285, "sampling/importance_sampling_ratio/mean": 1.0321862697601318, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3364130854606628, "clip_ratio/low_mean": 0.028233150951564312, "clip_ratio/low_min": 0.028233150951564312, "clip_ratio/high_mean": 0.08829259779304266, "clip_ratio/high_max": 0.08829259779304266, "clip_ratio/region_mean": 0.11652574874460697, "reward_total_mean": 0.5657984614372253, "reward_meter_mean": 0.7645368576049805, "reward_meter_std": 0.29501092433929443, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9716801643371582, "reward_repeat_soft_std": 0.014356398023664951, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.22947144508361816, "reward_total_composite_mean": 0.5657984614372253, "reward_total_composite_std": 0.2622303068637848} {"timestamp_utc": "2026-04-13T09:57:58Z", "mode": "train", "global_step": 1049, "epoch": 0.1053741838272225, "loss": -0.0369, "grad_norm": 5.236736297607422, "learning_rate": 6.824242424242425e-06, "num_tokens": 1854346.0, "completions/mean_length": 85.0, "completions/min_length": 17.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 24.000001907348633, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.7998566627502441, "rewards/meter/std": 0.3420889675617218, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9669623374938965, "rewards/repeat_soft/std": 0.0133640943095088, "rewards/judge_quality/mean": 0.41875001788139343, "rewards/judge_quality/std": 0.24398112297058105, "rewards/total_composite/mean": 0.4902343153953552, "rewards/total_composite/std": 0.324945330619812, "reward": 0.4902343153953552, "reward_std": 0.324945330619812, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17035700380802155, "sampling/sampling_logp_difference/max": 0.9106950759887695, "sampling/importance_sampling_ratio/min": 0.40224453806877136, "sampling/importance_sampling_ratio/mean": 1.0246261358261108, "sampling/importance_sampling_ratio/max": 1.9856619834899902, "entropy": 1.6026954054832458, "clip_ratio/low_mean": 0.01923076994717121, "clip_ratio/low_min": 0.01923076994717121, "clip_ratio/high_mean": 0.10076902061700821, "clip_ratio/high_max": 0.10076902061700821, "clip_ratio/region_mean": 0.11999979056417942, "reward_total_mean": 0.4902343153953552, "reward_meter_mean": 0.7998566627502441, "reward_meter_std": 0.3420889675617218, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9669623374938965, "reward_repeat_soft_std": 0.0133640943095088, "reward_judge_quality_mean": 0.41875001788139343, "reward_judge_quality_std": 0.24398112297058105, "reward_total_composite_mean": 0.4902343153953552, "reward_total_composite_std": 0.324945330619812} {"timestamp_utc": "2026-04-13T09:58:09Z", "mode": "train", "global_step": 1050, "epoch": 0.10547463586137619, "loss": -0.1394, "grad_norm": 1.9295040369033813, "learning_rate": 6.821212121212122e-06, "num_tokens": 1856068.0, "completions/mean_length": 172.25, "completions/min_length": 52.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 59.0, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.7838911414146423, "rewards/meter/std": 0.37247487902641296, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.946242094039917, "rewards/repeat_soft/std": 0.04032229632139206, "rewards/judge_quality/mean": 0.32749998569488525, "rewards/judge_quality/std": 0.17127670347690582, "rewards/total_composite/mean": 0.45562565326690674, "rewards/total_composite/std": 0.2813253104686737, "reward": 0.45562565326690674, "reward_std": 0.2813253104686737, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14240796864032745, "sampling/sampling_logp_difference/max": 1.0745148658752441, "sampling/importance_sampling_ratio/min": 0.3414633572101593, "sampling/importance_sampling_ratio/mean": 1.0431674718856812, "sampling/importance_sampling_ratio/max": 1.7962925434112549, "entropy": 1.164476990699768, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09937345143407583, "clip_ratio/high_max": 0.09937345143407583, "clip_ratio/region_mean": 0.09937345143407583, "reward_total_mean": 0.45562565326690674, "reward_meter_mean": 0.7838911414146423, "reward_meter_std": 0.37247487902641296, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.946242094039917, "reward_repeat_soft_std": 0.04032229632139206, "reward_judge_quality_mean": 0.32749998569488525, "reward_judge_quality_std": 0.17127670347690582, "reward_total_composite_mean": 0.45562565326690674, "reward_total_composite_std": 0.2813253104686737} {"timestamp_utc": "2026-04-13T09:59:00Z", "mode": "eval", "global_step": 1050, "epoch": 0.10547463586137619, "eval_loss": NaN, "eval_runtime": 50.6596, "eval_samples_per_second": 1.579, "eval_steps_per_second": 0.197, "eval_num_tokens": 1856068.0, "eval_completions/mean_length": 95.9, "eval_completions/min_length": 33.5, "eval_completions/max_length": 216.0, "eval_completions/clipped_ratio": 0.0875, "eval_completions/mean_terminated_length": 57.025000381469724, "eval_completions/min_terminated_length": 33.5, "eval_completions/max_terminated_length": 83.9, "eval_rewards/meter/mean": 0.6779823750257492, "eval_rewards/meter/std": 0.3323016606271267, "eval_rewards/count_adherence/mean": 0.8643750131130219, "eval_rewards/count_adherence/std": 0.14014482572674752, "eval_rewards/hard_gate/mean": 0.9125, "eval_rewards/hard_gate/std": 0.13509859144687653, "eval_rewards/repeat_soft/mean": 0.9586742341518402, "eval_rewards/repeat_soft/std": 0.043604453839361665, "eval_rewards/judge_quality/mean": 0.4231249958276749, "eval_rewards/judge_quality/std": 0.139389075525105, "eval_rewards/total_composite/mean": 0.48171039670705795, "eval_rewards/total_composite/std": 0.1573707213625312, "eval_reward": 0.48171039670705795, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.11468239948153496, "eval_sampling/sampling_logp_difference/max": 1.0133841991424561, "eval_sampling/importance_sampling_ratio/min": 0.38035872131586074, "eval_sampling/importance_sampling_ratio/mean": 1.0362835764884948, "eval_sampling/importance_sampling_ratio/max": 1.5530759930610656, "eval_entropy": 1.5260615825653077, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.48171039670705795, "eval_reward_meter_mean": 0.6779823750257492, "eval_reward_meter_std": 0.3323016606271267, "eval_reward_count_adherence_mean": 0.8643750131130219, "eval_reward_count_adherence_std": 0.14014482572674752, "eval_reward_hard_gate_mean": 0.9125, "eval_reward_hard_gate_std": 0.13509859144687653, "eval_reward_repeat_soft_mean": 0.9586742341518402, "eval_reward_repeat_soft_std": 0.043604453839361665, "eval_reward_judge_quality_mean": 0.4231249958276749, "eval_reward_judge_quality_std": 0.139389075525105, "eval_reward_total_composite_mean": 0.48171039670705795, "eval_reward_total_composite_std": 0.1573707213625312} {"timestamp_utc": "2026-04-13T09:59:14Z", "mode": "train", "global_step": 1051, "epoch": 0.10557508789552988, "loss": -0.0744, "grad_norm": 2.019414186477661, "learning_rate": 6.818181818181818e-06, "num_tokens": 1857417.0, "completions/mean_length": 83.625, "completions/min_length": 18.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 22.428571701049805, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 25.0, "rewards/meter/mean": 0.8272644281387329, "rewards/meter/std": 0.19974714517593384, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9372775554656982, "rewards/repeat_soft/std": 0.04599282518029213, "rewards/judge_quality/mean": 0.3387500047683716, "rewards/judge_quality/std": 0.14327171444892883, "rewards/total_composite/mean": 0.47729331254959106, "rewards/total_composite/std": 0.20679502189159393, "reward": 0.47729331254959106, "reward_std": 0.20679503679275513, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1740417629480362, "sampling/sampling_logp_difference/max": 1.1143274307250977, "sampling/importance_sampling_ratio/min": 0.3281359076499939, "sampling/importance_sampling_ratio/mean": 1.0323822498321533, "sampling/importance_sampling_ratio/max": 1.8156473636627197, "entropy": 1.4743000343441963, "clip_ratio/low_mean": 0.0243055559694767, "clip_ratio/low_min": 0.0243055559694767, "clip_ratio/high_mean": 0.10145257040858269, "clip_ratio/high_max": 0.10145257040858269, "clip_ratio/region_mean": 0.1257581263780594, "reward_total_mean": 0.47729331254959106, "reward_meter_mean": 0.8272644281387329, "reward_meter_std": 0.19974714517593384, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9372775554656982, "reward_repeat_soft_std": 0.04599282518029213, "reward_judge_quality_mean": 0.3387500047683716, "reward_judge_quality_std": 0.14327171444892883, "reward_total_composite_mean": 0.47729331254959106, "reward_total_composite_std": 0.20679502189159393} {"timestamp_utc": "2026-04-13T09:59:21Z", "mode": "train", "global_step": 1052, "epoch": 0.10567553992968358, "loss": 0.0115, "grad_norm": 10.280098915100098, "learning_rate": 6.8151515151515155e-06, "num_tokens": 1859368.0, "completions/mean_length": 66.875, "completions/min_length": 59.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.875, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9164806604385376, "rewards/meter/std": 0.20353005826473236, "rewards/count_adherence/mean": 0.6000000238418579, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9280893206596375, "rewards/repeat_soft/std": 0.04672547057271004, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.4956486225128174, "rewards/total_composite/std": 0.062345072627067566, "reward": 0.4956486225128174, "reward_std": 0.062345072627067566, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1618267148733139, "sampling/sampling_logp_difference/max": 1.3241872787475586, "sampling/importance_sampling_ratio/min": 0.2660190761089325, "sampling/importance_sampling_ratio/mean": 1.0189158916473389, "sampling/importance_sampling_ratio/max": 1.9141427278518677, "entropy": 1.711425855755806, "clip_ratio/low_mean": 0.037193973548710346, "clip_ratio/low_min": 0.037193973548710346, "clip_ratio/high_mean": 0.12622408755123615, "clip_ratio/high_max": 0.12622408755123615, "clip_ratio/region_mean": 0.1634180610999465, "reward_total_mean": 0.4956486225128174, "reward_meter_mean": 0.9164806604385376, "reward_meter_std": 0.20353005826473236, "reward_count_adherence_mean": 0.6000000238418579, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9280893206596375, "reward_repeat_soft_std": 0.04672547057271004, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.4956486225128174, "reward_total_composite_std": 0.062345072627067566} {"timestamp_utc": "2026-04-13T09:59:27Z", "mode": "train", "global_step": 1053, "epoch": 0.10577599196383727, "loss": -0.0086, "grad_norm": 10.646592140197754, "learning_rate": 6.812121212121212e-06, "num_tokens": 1861314.0, "completions/mean_length": 64.25, "completions/min_length": 58.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.25, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.7658095359802246, "rewards/meter/std": 0.3318875730037689, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.96340012550354, "rewards/repeat_soft/std": 0.017175309360027313, "rewards/judge_quality/mean": 0.29249998927116394, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.452426552772522, "rewards/total_composite/std": 0.08626030385494232, "reward": 0.452426552772522, "reward_std": 0.08626029640436172, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1762935072183609, "sampling/sampling_logp_difference/max": 1.2531490325927734, "sampling/importance_sampling_ratio/min": 0.28560400009155273, "sampling/importance_sampling_ratio/mean": 1.0375665426254272, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7278798073530197, "clip_ratio/low_mean": 0.06385537888854742, "clip_ratio/low_min": 0.06385537888854742, "clip_ratio/high_mean": 0.10123779810965061, "clip_ratio/high_max": 0.10123779810965061, "clip_ratio/region_mean": 0.16509317699819803, "reward_total_mean": 0.452426552772522, "reward_meter_mean": 0.7658095359802246, "reward_meter_std": 0.3318875730037689, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.96340012550354, "reward_repeat_soft_std": 0.017175309360027313, "reward_judge_quality_mean": 0.29249998927116394, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.452426552772522, "reward_total_composite_std": 0.08626030385494232} {"timestamp_utc": "2026-04-13T09:59:38Z", "mode": "train", "global_step": 1054, "epoch": 0.10587644399799095, "loss": -0.1456, "grad_norm": 3.234853506088257, "learning_rate": 6.80909090909091e-06, "num_tokens": 1863005.0, "completions/mean_length": 108.375, "completions/min_length": 41.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 50.71428680419922, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9629600048065186, "rewards/meter/std": 0.04580124840140343, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9219887852668762, "rewards/repeat_soft/std": 0.07961937040090561, "rewards/judge_quality/mean": 0.4725000262260437, "rewards/judge_quality/std": 0.24364788830280304, "rewards/total_composite/mean": 0.5899137258529663, "rewards/total_composite/std": 0.2688349187374115, "reward": 0.5899137258529663, "reward_std": 0.2688349187374115, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1691655069589615, "sampling/sampling_logp_difference/max": 1.1366252899169922, "sampling/importance_sampling_ratio/min": 0.3507244884967804, "sampling/importance_sampling_ratio/mean": 1.0380213260650635, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2046373188495636, "clip_ratio/low_mean": 0.029059450142085552, "clip_ratio/low_min": 0.029059450142085552, "clip_ratio/high_mean": 0.08967863582074642, "clip_ratio/high_max": 0.08967863582074642, "clip_ratio/region_mean": 0.11873808596283197, "reward_total_mean": 0.5899137258529663, "reward_meter_mean": 0.9629600048065186, "reward_meter_std": 0.04580124840140343, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9219887852668762, "reward_repeat_soft_std": 0.07961937040090561, "reward_judge_quality_mean": 0.4725000262260437, "reward_judge_quality_std": 0.24364788830280304, "reward_total_composite_mean": 0.5899137258529663, "reward_total_composite_std": 0.2688349187374115} {"timestamp_utc": "2026-04-13T09:59:44Z", "mode": "train", "global_step": 1055, "epoch": 0.10597689603214465, "loss": 0.1372, "grad_norm": 20.88136100769043, "learning_rate": 6.806060606060607e-06, "num_tokens": 1864564.0, "completions/mean_length": 34.875, "completions/min_length": 30.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.875, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.5309699773788452, "rewards/meter/std": 0.39381441473960876, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.992468535900116, "rewards/repeat_soft/std": 0.01208141166716814, "rewards/judge_quality/mean": 0.6850000023841858, "rewards/judge_quality/std": 0.21837061643600464, "rewards/total_composite/mean": 0.5953447818756104, "rewards/total_composite/std": 0.21855324506759644, "reward": 0.5953447818756104, "reward_std": 0.21855324506759644, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1475285142660141, "sampling/sampling_logp_difference/max": 1.246315360069275, "sampling/importance_sampling_ratio/min": 0.28756242990493774, "sampling/importance_sampling_ratio/mean": 1.0113372802734375, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9731053039431572, "clip_ratio/low_mean": 0.05729138432070613, "clip_ratio/low_min": 0.05729138432070613, "clip_ratio/high_mean": 0.06666379282251, "clip_ratio/high_max": 0.06666379282251, "clip_ratio/region_mean": 0.12395517714321613, "reward_total_mean": 0.5953447818756104, "reward_meter_mean": 0.5309699773788452, "reward_meter_std": 0.39381441473960876, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.992468535900116, "reward_repeat_soft_std": 0.01208141166716814, "reward_judge_quality_mean": 0.6850000023841858, "reward_judge_quality_std": 0.21837061643600464, "reward_total_composite_mean": 0.5953447818756104, "reward_total_composite_std": 0.21855324506759644} {"timestamp_utc": "2026-04-13T09:59:55Z", "mode": "train", "global_step": 1056, "epoch": 0.10607734806629834, "loss": -0.1162, "grad_norm": 3.0492923259735107, "learning_rate": 6.803030303030304e-06, "num_tokens": 1866155.0, "completions/mean_length": 103.875, "completions/min_length": 39.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 45.57143020629883, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.7495633363723755, "rewards/meter/std": 0.3189528286457062, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.15430334210395813, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9526692628860474, "rewards/repeat_soft/std": 0.07355757802724838, "rewards/judge_quality/mean": 0.5612500309944153, "rewards/judge_quality/std": 0.3223324716091156, "rewards/total_composite/mean": 0.5910525918006897, "rewards/total_composite/std": 0.2746005058288574, "reward": 0.5910525918006897, "reward_std": 0.2746005058288574, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14629122614860535, "sampling/sampling_logp_difference/max": 1.2436683177947998, "sampling/importance_sampling_ratio/min": 0.28832462430000305, "sampling/importance_sampling_ratio/mean": 1.023461937904358, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8375407457351685, "clip_ratio/low_mean": 0.040999338030815125, "clip_ratio/low_min": 0.040999338030815125, "clip_ratio/high_mean": 0.0836129542440176, "clip_ratio/high_max": 0.0836129542440176, "clip_ratio/region_mean": 0.12461229227483273, "reward_total_mean": 0.5910525918006897, "reward_meter_mean": 0.7495633363723755, "reward_meter_std": 0.3189528286457062, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.15430334210395813, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9526692628860474, "reward_repeat_soft_std": 0.07355757802724838, "reward_judge_quality_mean": 0.5612500309944153, "reward_judge_quality_std": 0.3223324716091156, "reward_total_composite_mean": 0.5910525918006897, "reward_total_composite_std": 0.2746005058288574} {"timestamp_utc": "2026-04-13T10:00:02Z", "mode": "train", "global_step": 1057, "epoch": 0.10617780010045204, "loss": -0.0128, "grad_norm": 11.504645347595215, "learning_rate": 6.800000000000001e-06, "num_tokens": 1867795.0, "completions/mean_length": 42.0, "completions/min_length": 38.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.0, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.5639481544494629, "rewards/meter/std": 0.3438780605792999, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9573047161102295, "rewards/repeat_soft/std": 0.0400361493229866, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.5008670091629028, "rewards/total_composite/std": 0.09752810746431351, "reward": 0.5008670091629028, "reward_std": 0.0975281149148941, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13496150076389313, "sampling/sampling_logp_difference/max": 2.762829303741455, "sampling/importance_sampling_ratio/min": 0.06311295181512833, "sampling/importance_sampling_ratio/mean": 1.0118708610534668, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0667807087302208, "clip_ratio/low_mean": 0.07114577945321798, "clip_ratio/low_min": 0.07114577945321798, "clip_ratio/high_mean": 0.0737397177144885, "clip_ratio/high_max": 0.0737397177144885, "clip_ratio/region_mean": 0.1448854971677065, "reward_total_mean": 0.5008670091629028, "reward_meter_mean": 0.5639481544494629, "reward_meter_std": 0.3438780605792999, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9573047161102295, "reward_repeat_soft_std": 0.0400361493229866, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.5008670091629028, "reward_total_composite_std": 0.09752810746431351} {"timestamp_utc": "2026-04-13T10:00:09Z", "mode": "train", "global_step": 1058, "epoch": 0.10627825213460572, "loss": -0.0093, "grad_norm": 10.222298622131348, "learning_rate": 6.796969696969697e-06, "num_tokens": 1870232.0, "completions/mean_length": 77.625, "completions/min_length": 59.0, "completions/max_length": 89.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.625, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 89.0, "rewards/meter/mean": 0.7992756366729736, "rewards/meter/std": 0.2670969069004059, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8280136585235596, "rewards/repeat_soft/std": 0.061355601996183395, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.4604276418685913, "rewards/total_composite/std": 0.09280714392662048, "reward": 0.4604276418685913, "reward_std": 0.0928071141242981, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13437646627426147, "sampling/sampling_logp_difference/max": 1.5356802940368652, "sampling/importance_sampling_ratio/min": 0.21530917286872864, "sampling/importance_sampling_ratio/mean": 0.9962149262428284, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0224271193146706, "clip_ratio/low_mean": 0.036021286621689796, "clip_ratio/low_min": 0.036021286621689796, "clip_ratio/high_mean": 0.08083718456327915, "clip_ratio/high_max": 0.08083718456327915, "clip_ratio/region_mean": 0.11685847118496895, "reward_total_mean": 0.4604276418685913, "reward_meter_mean": 0.7992756366729736, "reward_meter_std": 0.2670969069004059, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8280136585235596, "reward_repeat_soft_std": 0.061355601996183395, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.4604276418685913, "reward_total_composite_std": 0.09280714392662048} {"timestamp_utc": "2026-04-13T10:00:16Z", "mode": "train", "global_step": 1059, "epoch": 0.10637870416875941, "loss": -0.0395, "grad_norm": 18.02076530456543, "learning_rate": 6.793939393939395e-06, "num_tokens": 1871661.0, "completions/mean_length": 21.625, "completions/min_length": 18.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.625, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.7551445960998535, "rewards/meter/std": 0.3369503915309906, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9593343734741211, "rewards/repeat_soft/std": 0.008953613229095936, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.5632452964782715, "rewards/total_composite/std": 0.08104057610034943, "reward": 0.5632452964782715, "reward_std": 0.08104058355093002, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15181419253349304, "sampling/sampling_logp_difference/max": 0.9656963348388672, "sampling/importance_sampling_ratio/min": 0.38071802258491516, "sampling/importance_sampling_ratio/mean": 1.0368188619613647, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3528812900185585, "clip_ratio/low_mean": 0.040570175275206566, "clip_ratio/low_min": 0.040570175275206566, "clip_ratio/high_mean": 0.10240816045552492, "clip_ratio/high_max": 0.10240816045552492, "clip_ratio/region_mean": 0.1429783357307315, "reward_total_mean": 0.5632452964782715, "reward_meter_mean": 0.7551445960998535, "reward_meter_std": 0.3369503915309906, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9593343734741211, "reward_repeat_soft_std": 0.008953613229095936, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.5632452964782715, "reward_total_composite_std": 0.08104057610034943} {"timestamp_utc": "2026-04-13T10:00:22Z", "mode": "train", "global_step": 1060, "epoch": 0.10647915620291311, "loss": 0.0055, "grad_norm": 10.958192825317383, "learning_rate": 6.790909090909091e-06, "num_tokens": 1873690.0, "completions/mean_length": 73.625, "completions/min_length": 67.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.625, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.8756072521209717, "rewards/meter/std": 0.17839808762073517, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9417939186096191, "rewards/repeat_soft/std": 0.03722754493355751, "rewards/judge_quality/mean": 0.5275000333786011, "rewards/judge_quality/std": 0.18873640894889832, "rewards/total_composite/mean": 0.5927358865737915, "rewards/total_composite/std": 0.12903288006782532, "reward": 0.5927358865737915, "reward_std": 0.12903286516666412, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1629772037267685, "sampling/sampling_logp_difference/max": 3.0517749786376953, "sampling/importance_sampling_ratio/min": 0.047274935990571976, "sampling/importance_sampling_ratio/mean": 1.0284122228622437, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.303021028637886, "clip_ratio/low_mean": 0.09073344804346561, "clip_ratio/low_min": 0.09073344804346561, "clip_ratio/high_mean": 0.03935917653143406, "clip_ratio/high_max": 0.03935917653143406, "clip_ratio/region_mean": 0.13009262457489967, "reward_total_mean": 0.5927358865737915, "reward_meter_mean": 0.8756072521209717, "reward_meter_std": 0.17839808762073517, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9417939186096191, "reward_repeat_soft_std": 0.03722754493355751, "reward_judge_quality_mean": 0.5275000333786011, "reward_judge_quality_std": 0.18873640894889832, "reward_total_composite_mean": 0.5927358865737915, "reward_total_composite_std": 0.12903288006782532} {"timestamp_utc": "2026-04-13T10:00:28Z", "mode": "train", "global_step": 1061, "epoch": 0.1065796082370668, "loss": 0.0683, "grad_norm": 12.164383888244629, "learning_rate": 6.787878787878789e-06, "num_tokens": 1875233.0, "completions/mean_length": 43.875, "completions/min_length": 34.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.875, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9468818306922913, "rewards/meter/std": 0.04013519361615181, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9603518843650818, "rewards/repeat_soft/std": 0.029111694544553757, "rewards/judge_quality/mean": 0.6112500429153442, "rewards/judge_quality/std": 0.25587037205696106, "rewards/total_composite/mean": 0.7183948159217834, "rewards/total_composite/std": 0.15400682389736176, "reward": 0.7183948159217834, "reward_std": 0.15400682389736176, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13824084401130676, "sampling/sampling_logp_difference/max": 1.3404159545898438, "sampling/importance_sampling_ratio/min": 0.26173678040504456, "sampling/importance_sampling_ratio/mean": 1.019260287284851, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2784786894917488, "clip_ratio/low_mean": 0.06544010294601321, "clip_ratio/low_min": 0.06544010294601321, "clip_ratio/high_mean": 0.024099512491375208, "clip_ratio/high_max": 0.024099512491375208, "clip_ratio/region_mean": 0.08953961543738842, "reward_total_mean": 0.7183948159217834, "reward_meter_mean": 0.9468818306922913, "reward_meter_std": 0.04013519361615181, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9603518843650818, "reward_repeat_soft_std": 0.029111694544553757, "reward_judge_quality_mean": 0.6112500429153442, "reward_judge_quality_std": 0.25587037205696106, "reward_total_composite_mean": 0.7183948159217834, "reward_total_composite_std": 0.15400682389736176} {"timestamp_utc": "2026-04-13T10:00:34Z", "mode": "train", "global_step": 1062, "epoch": 0.1066800602712205, "loss": 0.0408, "grad_norm": 10.422037124633789, "learning_rate": 6.7848484848484855e-06, "num_tokens": 1877171.0, "completions/mean_length": 67.25, "completions/min_length": 61.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.25, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.971000075340271, "rewards/meter/std": 0.0235135480761528, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9678769111633301, "rewards/repeat_soft/std": 0.014561736956238747, "rewards/judge_quality/mean": 0.4362499713897705, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.5703381896018982, "rewards/total_composite/std": 0.08263786137104034, "reward": 0.5703381896018982, "reward_std": 0.08263785392045975, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1390233039855957, "sampling/sampling_logp_difference/max": 1.2323393821716309, "sampling/importance_sampling_ratio/min": 0.29160958528518677, "sampling/importance_sampling_ratio/mean": 1.012794852256775, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1702293306589127, "clip_ratio/low_mean": 0.10602645669132471, "clip_ratio/low_min": 0.10602645669132471, "clip_ratio/high_mean": 0.02459016442298889, "clip_ratio/high_max": 0.02459016442298889, "clip_ratio/region_mean": 0.1306166211143136, "reward_total_mean": 0.5703381896018982, "reward_meter_mean": 0.971000075340271, "reward_meter_std": 0.0235135480761528, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9678769111633301, "reward_repeat_soft_std": 0.014561736956238747, "reward_judge_quality_mean": 0.4362499713897705, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.5703381896018982, "reward_total_composite_std": 0.08263786137104034} {"timestamp_utc": "2026-04-13T10:00:41Z", "mode": "train", "global_step": 1063, "epoch": 0.10678051230537418, "loss": 0.0279, "grad_norm": 13.100746154785156, "learning_rate": 6.781818181818183e-06, "num_tokens": 1879210.0, "completions/mean_length": 73.875, "completions/min_length": 65.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.875, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.8981308937072754, "rewards/meter/std": 0.23610873520374298, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9531311988830566, "rewards/repeat_soft/std": 0.042662639170885086, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5381593704223633, "rewards/total_composite/std": 0.06434082239866257, "reward": 0.5381593704223633, "reward_std": 0.06434082239866257, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15877306461334229, "sampling/sampling_logp_difference/max": 2.6850342750549316, "sampling/importance_sampling_ratio/min": 0.23416869342327118, "sampling/importance_sampling_ratio/mean": 1.0295064449310303, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2262151837348938, "clip_ratio/low_mean": 0.036093054339289665, "clip_ratio/low_min": 0.036093054339289665, "clip_ratio/high_mean": 0.10448102559894323, "clip_ratio/high_max": 0.10448102559894323, "clip_ratio/region_mean": 0.1405740799382329, "reward_total_mean": 0.5381593704223633, "reward_meter_mean": 0.8981308937072754, "reward_meter_std": 0.23610873520374298, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9531311988830566, "reward_repeat_soft_std": 0.042662639170885086, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5381593704223633, "reward_total_composite_std": 0.06434082239866257} {"timestamp_utc": "2026-04-13T10:00:49Z", "mode": "train", "global_step": 1064, "epoch": 0.10688096433952787, "loss": -0.0478, "grad_norm": 18.498001098632812, "learning_rate": 6.778787878787879e-06, "num_tokens": 1880747.0, "completions/mean_length": 23.125, "completions/min_length": 18.0, "completions/max_length": 27.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.125, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.9713336229324341, "rewards/meter/std": 0.020421607419848442, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9412878751754761, "rewards/repeat_soft/std": 0.04843747243285179, "rewards/judge_quality/mean": 0.581250011920929, "rewards/judge_quality/std": 0.2968134582042694, "rewards/total_composite/mean": 0.7095893621444702, "rewards/total_composite/std": 0.18525193631649017, "reward": 0.7095893621444702, "reward_std": 0.18525192141532898, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19308802485466003, "sampling/sampling_logp_difference/max": 1.3790791034698486, "sampling/importance_sampling_ratio/min": 0.25181034207344055, "sampling/importance_sampling_ratio/mean": 1.010938048362732, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4194799214601517, "clip_ratio/low_mean": 0.0904420162551105, "clip_ratio/low_min": 0.0904420162551105, "clip_ratio/high_mean": 0.03762464504688978, "clip_ratio/high_max": 0.03762464504688978, "clip_ratio/region_mean": 0.12806666130200028, "reward_total_mean": 0.7095893621444702, "reward_meter_mean": 0.9713336229324341, "reward_meter_std": 0.020421607419848442, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9412878751754761, "reward_repeat_soft_std": 0.04843747243285179, "reward_judge_quality_mean": 0.581250011920929, "reward_judge_quality_std": 0.2968134582042694, "reward_total_composite_mean": 0.7095893621444702, "reward_total_composite_std": 0.18525193631649017} {"timestamp_utc": "2026-04-13T10:00:55Z", "mode": "train", "global_step": 1065, "epoch": 0.10698141637368157, "loss": 0.107, "grad_norm": 10.864692687988281, "learning_rate": 6.7757575757575765e-06, "num_tokens": 1882443.0, "completions/mean_length": 45.0, "completions/min_length": 36.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.0, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9339120388031006, "rewards/meter/std": 0.1331321746110916, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9318022131919861, "rewards/repeat_soft/std": 0.04915972426533699, "rewards/judge_quality/mean": 0.4024999737739563, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.5836708545684814, "rewards/total_composite/std": 0.0491509884595871, "reward": 0.5836708545684814, "reward_std": 0.049150995910167694, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16631630063056946, "sampling/sampling_logp_difference/max": 1.5867557525634766, "sampling/importance_sampling_ratio/min": 0.20458826422691345, "sampling/importance_sampling_ratio/mean": 1.01224684715271, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.440430074930191, "clip_ratio/low_mean": 0.04046474304050207, "clip_ratio/low_min": 0.04046474304050207, "clip_ratio/high_mean": 0.09686570474877954, "clip_ratio/high_max": 0.09686570474877954, "clip_ratio/region_mean": 0.1373304477892816, "reward_total_mean": 0.5836708545684814, "reward_meter_mean": 0.9339120388031006, "reward_meter_std": 0.1331321746110916, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9318022131919861, "reward_repeat_soft_std": 0.04915972426533699, "reward_judge_quality_mean": 0.4024999737739563, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.5836708545684814, "reward_total_composite_std": 0.0491509884595871} {"timestamp_utc": "2026-04-13T10:01:10Z", "mode": "train", "global_step": 1066, "epoch": 0.10708186840783526, "loss": -0.1539, "grad_norm": 2.0058774948120117, "learning_rate": 6.772727272727273e-06, "num_tokens": 1884430.0, "completions/mean_length": 119.375, "completions/min_length": 50.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 63.28571701049805, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.9014915227890015, "rewards/meter/std": 0.13286340236663818, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9050803184509277, "rewards/repeat_soft/std": 0.05679691955447197, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.13593590259552002, "rewards/total_composite/mean": 0.4573853015899658, "rewards/total_composite/std": 0.19074977934360504, "reward": 0.4573853015899658, "reward_std": 0.19074977934360504, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11312473565340042, "sampling/sampling_logp_difference/max": 1.3569793701171875, "sampling/importance_sampling_ratio/min": 0.25743722915649414, "sampling/importance_sampling_ratio/mean": 1.0099263191223145, "sampling/importance_sampling_ratio/max": 1.7764612436294556, "entropy": 0.8119281008839607, "clip_ratio/low_mean": 0.009803921915590763, "clip_ratio/low_min": 0.009803921915590763, "clip_ratio/high_mean": 0.07999108778312802, "clip_ratio/high_max": 0.07999108778312802, "clip_ratio/region_mean": 0.08979500969871879, "reward_total_mean": 0.4573853015899658, "reward_meter_mean": 0.9014915227890015, "reward_meter_std": 0.13286340236663818, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9050803184509277, "reward_repeat_soft_std": 0.05679691955447197, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.13593590259552002, "reward_total_composite_mean": 0.4573853015899658, "reward_total_composite_std": 0.19074977934360504} {"timestamp_utc": "2026-04-13T10:01:16Z", "mode": "train", "global_step": 1067, "epoch": 0.10718232044198896, "loss": 0.0287, "grad_norm": 16.06044578552246, "learning_rate": 6.76969696969697e-06, "num_tokens": 1885995.0, "completions/mean_length": 36.625, "completions/min_length": 32.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.625, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.8607951402664185, "rewards/meter/std": 0.34211716055870056, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9544247984886169, "rewards/repeat_soft/std": 0.0466926284134388, "rewards/judge_quality/mean": 0.5862500071525574, "rewards/judge_quality/std": 0.22984081506729126, "rewards/total_composite/mean": 0.6810033321380615, "rewards/total_composite/std": 0.19374404847621918, "reward": 0.6810033321380615, "reward_std": 0.19374403357505798, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15902628004550934, "sampling/sampling_logp_difference/max": 2.501688003540039, "sampling/importance_sampling_ratio/min": 0.08194655179977417, "sampling/importance_sampling_ratio/mean": 0.9926603436470032, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8786809593439102, "clip_ratio/low_mean": 0.1110491082072258, "clip_ratio/low_min": 0.1110491082072258, "clip_ratio/high_mean": 0.06022810656577349, "clip_ratio/high_max": 0.06022810656577349, "clip_ratio/region_mean": 0.1712772147729993, "reward_total_mean": 0.6810033321380615, "reward_meter_mean": 0.8607951402664185, "reward_meter_std": 0.34211716055870056, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9544247984886169, "reward_repeat_soft_std": 0.0466926284134388, "reward_judge_quality_mean": 0.5862500071525574, "reward_judge_quality_std": 0.22984081506729126, "reward_total_composite_mean": 0.6810033321380615, "reward_total_composite_std": 0.19374404847621918} {"timestamp_utc": "2026-04-13T10:01:22Z", "mode": "train", "global_step": 1068, "epoch": 0.10728277247614264, "loss": -0.023, "grad_norm": 12.230772972106934, "learning_rate": 6.7666666666666665e-06, "num_tokens": 1887856.0, "completions/mean_length": 72.625, "completions/min_length": 63.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.625, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9264994859695435, "rewards/meter/std": 0.0505947582423687, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9661044478416443, "rewards/repeat_soft/std": 0.01599237136542797, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5478500127792358, "rewards/total_composite/std": 0.014911556616425514, "reward": 0.5478500127792358, "reward_std": 0.014911546371877193, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12288069725036621, "sampling/sampling_logp_difference/max": 0.9862861633300781, "sampling/importance_sampling_ratio/min": 0.3729592263698578, "sampling/importance_sampling_ratio/mean": 1.034889817237854, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0377146005630493, "clip_ratio/low_mean": 0.0284205237403512, "clip_ratio/low_min": 0.0284205237403512, "clip_ratio/high_mean": 0.08064270671457052, "clip_ratio/high_max": 0.08064270671457052, "clip_ratio/region_mean": 0.10906323045492172, "reward_total_mean": 0.5478500127792358, "reward_meter_mean": 0.9264994859695435, "reward_meter_std": 0.0505947582423687, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9661044478416443, "reward_repeat_soft_std": 0.01599237136542797, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5478500127792358, "reward_total_composite_std": 0.014911556616425514} {"timestamp_utc": "2026-04-13T10:01:28Z", "mode": "train", "global_step": 1069, "epoch": 0.10738322451029633, "loss": 0.001, "grad_norm": 14.955061912536621, "learning_rate": 6.763636363636365e-06, "num_tokens": 1889477.0, "completions/mean_length": 47.625, "completions/min_length": 43.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.625, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.7776228189468384, "rewards/meter/std": 0.27847403287887573, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9883979558944702, "rewards/repeat_soft/std": 0.013899828307330608, "rewards/judge_quality/mean": 0.8050000071525574, "rewards/judge_quality/std": 0.16860775649547577, "rewards/total_composite/mean": 0.7516942620277405, "rewards/total_composite/std": 0.16910621523857117, "reward": 0.7516942620277405, "reward_std": 0.16910621523857117, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11631647497415543, "sampling/sampling_logp_difference/max": 2.1365227699279785, "sampling/importance_sampling_ratio/min": 0.11806467175483704, "sampling/importance_sampling_ratio/mean": 1.0089021921157837, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5627031028270721, "clip_ratio/low_mean": 0.032379837008193135, "clip_ratio/low_min": 0.032379837008193135, "clip_ratio/high_mean": 0.08231146773323417, "clip_ratio/high_max": 0.08231146773323417, "clip_ratio/region_mean": 0.1146913047414273, "reward_total_mean": 0.7516942620277405, "reward_meter_mean": 0.7776228189468384, "reward_meter_std": 0.27847403287887573, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9883979558944702, "reward_repeat_soft_std": 0.013899828307330608, "reward_judge_quality_mean": 0.8050000071525574, "reward_judge_quality_std": 0.16860775649547577, "reward_total_composite_mean": 0.7516942620277405, "reward_total_composite_std": 0.16910621523857117} {"timestamp_utc": "2026-04-13T10:01:34Z", "mode": "train", "global_step": 1070, "epoch": 0.10748367654445003, "loss": 0.0614, "grad_norm": 11.298134803771973, "learning_rate": 6.760606060606061e-06, "num_tokens": 1891061.0, "completions/mean_length": 41.0, "completions/min_length": 35.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9851936101913452, "rewards/meter/std": 0.007341459859162569, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9215592741966248, "rewards/repeat_soft/std": 0.05847397819161415, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.6871336698532104, "rewards/total_composite/std": 0.14895614981651306, "reward": 0.6871336698532104, "reward_std": 0.14895614981651306, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1290155053138733, "sampling/sampling_logp_difference/max": 1.2429022789001465, "sampling/importance_sampling_ratio/min": 0.2885455787181854, "sampling/importance_sampling_ratio/mean": 1.0117584466934204, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8554912656545639, "clip_ratio/low_mean": 0.0737000578083098, "clip_ratio/low_min": 0.0737000578083098, "clip_ratio/high_mean": 0.03482142882421613, "clip_ratio/high_max": 0.03482142882421613, "clip_ratio/region_mean": 0.10852148663252592, "reward_total_mean": 0.6871336698532104, "reward_meter_mean": 0.9851936101913452, "reward_meter_std": 0.007341459859162569, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9215592741966248, "reward_repeat_soft_std": 0.05847397819161415, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.6871336698532104, "reward_total_composite_std": 0.14895614981651306} {"timestamp_utc": "2026-04-13T10:01:41Z", "mode": "train", "global_step": 1071, "epoch": 0.10758412857860372, "loss": 0.0196, "grad_norm": 10.647320747375488, "learning_rate": 6.757575757575758e-06, "num_tokens": 1893169.0, "completions/mean_length": 67.5, "completions/min_length": 59.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.5, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.9314833283424377, "rewards/meter/std": 0.09214688092470169, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9345080852508545, "rewards/repeat_soft/std": 0.05088561773300171, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.10966669768095016, "rewards/total_composite/mean": 0.4987291395664215, "rewards/total_composite/std": 0.06691356748342514, "reward": 0.4987291395664215, "reward_std": 0.06691356748342514, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14687930047512054, "sampling/sampling_logp_difference/max": 2.208160877227783, "sampling/importance_sampling_ratio/min": 0.10990258306264877, "sampling/importance_sampling_ratio/mean": 1.0051355361938477, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2108322158455849, "clip_ratio/low_mean": 0.06116915214806795, "clip_ratio/low_min": 0.06116915214806795, "clip_ratio/high_mean": 0.07562159188091755, "clip_ratio/high_max": 0.07562159188091755, "clip_ratio/region_mean": 0.1367907440289855, "reward_total_mean": 0.4987291395664215, "reward_meter_mean": 0.9314833283424377, "reward_meter_std": 0.09214688092470169, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9345080852508545, "reward_repeat_soft_std": 0.05088561773300171, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.10966669768095016, "reward_total_composite_mean": 0.4987291395664215, "reward_total_composite_std": 0.06691356748342514} {"timestamp_utc": "2026-04-13T10:01:47Z", "mode": "train", "global_step": 1072, "epoch": 0.1076845806127574, "loss": 0.0086, "grad_norm": 16.221986770629883, "learning_rate": 6.754545454545455e-06, "num_tokens": 1894673.0, "completions/mean_length": 44.0, "completions/min_length": 42.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.0, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.8085959553718567, "rewards/meter/std": 0.262198805809021, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9473362565040588, "rewards/repeat_soft/std": 0.04573459178209305, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.5633194446563721, "rewards/total_composite/std": 0.06888439506292343, "reward": 0.5633194446563721, "reward_std": 0.06888439506292343, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11900659650564194, "sampling/sampling_logp_difference/max": 0.989473819732666, "sampling/importance_sampling_ratio/min": 0.4080321788787842, "sampling/importance_sampling_ratio/mean": 1.03138267993927, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0347795113921165, "clip_ratio/low_mean": 0.05752122774720192, "clip_ratio/low_min": 0.05752122774720192, "clip_ratio/high_mean": 0.06697582267224789, "clip_ratio/high_max": 0.06697582267224789, "clip_ratio/region_mean": 0.1244970504194498, "reward_total_mean": 0.5633194446563721, "reward_meter_mean": 0.8085959553718567, "reward_meter_std": 0.262198805809021, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9473362565040588, "reward_repeat_soft_std": 0.04573459178209305, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.5633194446563721, "reward_total_composite_std": 0.06888439506292343} {"timestamp_utc": "2026-04-13T10:01:53Z", "mode": "train", "global_step": 1073, "epoch": 0.1077850326469111, "loss": 0.0238, "grad_norm": 14.687736511230469, "learning_rate": 6.751515151515152e-06, "num_tokens": 1896182.0, "completions/mean_length": 42.625, "completions/min_length": 32.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.625, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.870096743106842, "rewards/meter/std": 0.1931677907705307, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.984889805316925, "rewards/repeat_soft/std": 0.010807708837091923, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.5965648293495178, "rewards/total_composite/std": 0.05712573602795601, "reward": 0.5965648293495178, "reward_std": 0.05712572857737541, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16201752424240112, "sampling/sampling_logp_difference/max": 1.2331657409667969, "sampling/importance_sampling_ratio/min": 0.2913687229156494, "sampling/importance_sampling_ratio/mean": 1.0427515506744385, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.352506972849369, "clip_ratio/low_mean": 0.06444899551570415, "clip_ratio/low_min": 0.06444899551570415, "clip_ratio/high_mean": 0.08135287696495652, "clip_ratio/high_max": 0.08135287696495652, "clip_ratio/region_mean": 0.14580187248066068, "reward_total_mean": 0.5965648293495178, "reward_meter_mean": 0.870096743106842, "reward_meter_std": 0.1931677907705307, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.984889805316925, "reward_repeat_soft_std": 0.010807708837091923, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.5965648293495178, "reward_total_composite_std": 0.05712573602795601} {"timestamp_utc": "2026-04-13T10:01:59Z", "mode": "train", "global_step": 1074, "epoch": 0.10788548468106479, "loss": 0.0522, "grad_norm": 12.577512741088867, "learning_rate": 6.748484848484848e-06, "num_tokens": 1897835.0, "completions/mean_length": 46.625, "completions/min_length": 35.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.625, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.6758618354797363, "rewards/meter/std": 0.2913423776626587, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9415541291236877, "rewards/repeat_soft/std": 0.04388896003365517, "rewards/judge_quality/mean": 0.6237499713897705, "rewards/judge_quality/std": 0.22890658676624298, "rewards/total_composite/mean": 0.6123546361923218, "rewards/total_composite/std": 0.1519799530506134, "reward": 0.6123546361923218, "reward_std": 0.1519799530506134, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12512366473674774, "sampling/sampling_logp_difference/max": 1.4607925415039062, "sampling/importance_sampling_ratio/min": 0.2320522964000702, "sampling/importance_sampling_ratio/mean": 1.0119976997375488, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6220041923224926, "clip_ratio/low_mean": 0.07702832575887442, "clip_ratio/low_min": 0.07702832575887442, "clip_ratio/high_mean": 0.059330222196877, "clip_ratio/high_max": 0.059330222196877, "clip_ratio/region_mean": 0.13635854795575142, "reward_total_mean": 0.6123546361923218, "reward_meter_mean": 0.6758618354797363, "reward_meter_std": 0.2913423776626587, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9415541291236877, "reward_repeat_soft_std": 0.04388896003365517, "reward_judge_quality_mean": 0.6237499713897705, "reward_judge_quality_std": 0.22890658676624298, "reward_total_composite_mean": 0.6123546361923218, "reward_total_composite_std": 0.1519799530506134} {"timestamp_utc": "2026-04-13T10:02:05Z", "mode": "train", "global_step": 1075, "epoch": 0.10798593671521849, "loss": 0.0121, "grad_norm": 27.427839279174805, "learning_rate": 6.7454545454545465e-06, "num_tokens": 1899412.0, "completions/mean_length": 20.125, "completions/min_length": 17.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 20.125, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.8380264043807983, "rewards/meter/std": 0.2947549521923065, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8501101732254028, "rewards/repeat_soft/std": 0.07873211055994034, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.0843462198972702, "rewards/total_composite/mean": 0.5350409746170044, "rewards/total_composite/std": 0.08704709261655807, "reward": 0.5350409746170044, "reward_std": 0.08704709261655807, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14486834406852722, "sampling/sampling_logp_difference/max": 1.1462392807006836, "sampling/importance_sampling_ratio/min": 0.31782978773117065, "sampling/importance_sampling_ratio/mean": 1.034041166305542, "sampling/importance_sampling_ratio/max": 1.9090057611465454, "entropy": 1.1850376725196838, "clip_ratio/low_mean": 0.03128482960164547, "clip_ratio/low_min": 0.03128482960164547, "clip_ratio/high_mean": 0.08520944183692336, "clip_ratio/high_max": 0.08520944183692336, "clip_ratio/region_mean": 0.11649427143856883, "reward_total_mean": 0.5350409746170044, "reward_meter_mean": 0.8380264043807983, "reward_meter_std": 0.2947549521923065, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8501101732254028, "reward_repeat_soft_std": 0.07873211055994034, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.0843462198972702, "reward_total_composite_mean": 0.5350409746170044, "reward_total_composite_std": 0.08704709261655807} {"timestamp_utc": "2026-04-13T10:02:12Z", "mode": "train", "global_step": 1076, "epoch": 0.10808638874937218, "loss": 0.0427, "grad_norm": 12.953285217285156, "learning_rate": 6.742424242424243e-06, "num_tokens": 1900961.0, "completions/mean_length": 41.625, "completions/min_length": 32.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.625, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.9870993494987488, "rewards/meter/std": 0.0081932432949543, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.932524561882019, "rewards/repeat_soft/std": 0.06071160361170769, "rewards/judge_quality/mean": 0.38875001668930054, "rewards/judge_quality/std": 0.08675704896450043, "rewards/total_composite/mean": 0.5891315340995789, "rewards/total_composite/std": 0.05717764049768448, "reward": 0.5891315340995789, "reward_std": 0.05717764422297478, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1406351923942566, "sampling/sampling_logp_difference/max": 1.5530462265014648, "sampling/importance_sampling_ratio/min": 0.21160240471363068, "sampling/importance_sampling_ratio/mean": 1.0404138565063477, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1810264438390732, "clip_ratio/low_mean": 0.03371212258934975, "clip_ratio/low_min": 0.03371212258934975, "clip_ratio/high_mean": 0.10204389691352844, "clip_ratio/high_max": 0.10204389691352844, "clip_ratio/region_mean": 0.1357560195028782, "reward_total_mean": 0.5891315340995789, "reward_meter_mean": 0.9870993494987488, "reward_meter_std": 0.0081932432949543, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.932524561882019, "reward_repeat_soft_std": 0.06071160361170769, "reward_judge_quality_mean": 0.38875001668930054, "reward_judge_quality_std": 0.08675704896450043, "reward_total_composite_mean": 0.5891315340995789, "reward_total_composite_std": 0.05717764049768448} {"timestamp_utc": "2026-04-13T10:02:24Z", "mode": "train", "global_step": 1077, "epoch": 0.10818684078352586, "loss": -0.1388, "grad_norm": 2.1381711959838867, "learning_rate": 6.73939393939394e-06, "num_tokens": 1902642.0, "completions/mean_length": 114.125, "completions/min_length": 44.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 57.28571701049805, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9222038984298706, "rewards/meter/std": 0.1504916250705719, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.976231575012207, "rewards/repeat_soft/std": 0.02974354662001133, "rewards/judge_quality/mean": 0.4150000214576721, "rewards/judge_quality/std": 0.18031719326972961, "rewards/total_composite/mean": 0.5495796799659729, "rewards/total_composite/std": 0.23759342730045319, "reward": 0.5495796799659729, "reward_std": 0.23759342730045319, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1703491061925888, "sampling/sampling_logp_difference/max": 1.1962089538574219, "sampling/importance_sampling_ratio/min": 0.30233821272850037, "sampling/importance_sampling_ratio/mean": 1.0201829671859741, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4231328517198563, "clip_ratio/low_mean": 0.01785714365541935, "clip_ratio/low_min": 0.01785714365541935, "clip_ratio/high_mean": 0.14230037666857243, "clip_ratio/high_max": 0.14230037666857243, "clip_ratio/region_mean": 0.16015752032399178, "reward_total_mean": 0.5495796799659729, "reward_meter_mean": 0.9222038984298706, "reward_meter_std": 0.1504916250705719, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.976231575012207, "reward_repeat_soft_std": 0.02974354662001133, "reward_judge_quality_mean": 0.4150000214576721, "reward_judge_quality_std": 0.18031719326972961, "reward_total_composite_mean": 0.5495796799659729, "reward_total_composite_std": 0.23759342730045319} {"timestamp_utc": "2026-04-13T10:02:35Z", "mode": "train", "global_step": 1078, "epoch": 0.10828729281767956, "loss": -0.1107, "grad_norm": 3.0935230255126953, "learning_rate": 6.7363636363636365e-06, "num_tokens": 1904408.0, "completions/mean_length": 102.75, "completions/min_length": 39.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 44.28571701049805, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.8180112838745117, "rewards/meter/std": 0.3261679708957672, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9903794527053833, "rewards/repeat_soft/std": 0.008472509682178497, "rewards/judge_quality/mean": 0.45249998569488525, "rewards/judge_quality/std": 0.21625052392482758, "rewards/total_composite/mean": 0.5496870875358582, "rewards/total_composite/std": 0.2644262909889221, "reward": 0.5496870875358582, "reward_std": 0.2644262909889221, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1494397521018982, "sampling/sampling_logp_difference/max": 1.6385045051574707, "sampling/importance_sampling_ratio/min": 0.19427035748958588, "sampling/importance_sampling_ratio/mean": 1.0164092779159546, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.141548976302147, "clip_ratio/low_mean": 0.016025641933083534, "clip_ratio/low_min": 0.016025641933083534, "clip_ratio/high_mean": 0.10979404021054506, "clip_ratio/high_max": 0.10979404021054506, "clip_ratio/region_mean": 0.1258196821436286, "reward_total_mean": 0.5496870875358582, "reward_meter_mean": 0.8180112838745117, "reward_meter_std": 0.3261679708957672, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9903794527053833, "reward_repeat_soft_std": 0.008472509682178497, "reward_judge_quality_mean": 0.45249998569488525, "reward_judge_quality_std": 0.21625052392482758, "reward_total_composite_mean": 0.5496870875358582, "reward_total_composite_std": 0.2644262909889221} {"timestamp_utc": "2026-04-13T10:02:42Z", "mode": "train", "global_step": 1079, "epoch": 0.10838774485183325, "loss": 0.0086, "grad_norm": 10.065486907958984, "learning_rate": 6.733333333333334e-06, "num_tokens": 1906543.0, "completions/mean_length": 85.875, "completions/min_length": 60.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.875, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.7328099608421326, "rewards/meter/std": 0.28351694345474243, "rewards/count_adherence/mean": 0.7749999761581421, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9198435544967651, "rewards/repeat_soft/std": 0.05026472732424736, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.504819929599762, "rewards/total_composite/std": 0.12204664200544357, "reward": 0.504819929599762, "reward_std": 0.12204664945602417, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14232237637043, "sampling/sampling_logp_difference/max": 1.5423803329467773, "sampling/importance_sampling_ratio/min": 0.21387141942977905, "sampling/importance_sampling_ratio/mean": 1.0154612064361572, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1735976114869118, "clip_ratio/low_mean": 0.06142867170274258, "clip_ratio/low_min": 0.06142867170274258, "clip_ratio/high_mean": 0.05976541340351105, "clip_ratio/high_max": 0.05976541340351105, "clip_ratio/region_mean": 0.12119408510625362, "reward_total_mean": 0.504819929599762, "reward_meter_mean": 0.7328099608421326, "reward_meter_std": 0.28351694345474243, "reward_count_adherence_mean": 0.7749999761581421, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9198435544967651, "reward_repeat_soft_std": 0.05026472732424736, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.504819929599762, "reward_total_composite_std": 0.12204664200544357} {"timestamp_utc": "2026-04-13T10:02:48Z", "mode": "train", "global_step": 1080, "epoch": 0.10848819688598695, "loss": 0.0736, "grad_norm": 16.76507568359375, "learning_rate": 6.73030303030303e-06, "num_tokens": 1907997.0, "completions/mean_length": 37.75, "completions/min_length": 35.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.75, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.7624353170394897, "rewards/meter/std": 0.41185691952705383, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9511844515800476, "rewards/repeat_soft/std": 0.039821553975343704, "rewards/judge_quality/mean": 0.8737500309944153, "rewards/judge_quality/std": 0.08601287752389908, "rewards/total_composite/mean": 0.7842920422554016, "rewards/total_composite/std": 0.24220286309719086, "reward": 0.7842920422554016, "reward_std": 0.24220287799835205, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13939060270786285, "sampling/sampling_logp_difference/max": 1.7247705459594727, "sampling/importance_sampling_ratio/min": 0.1782139390707016, "sampling/importance_sampling_ratio/mean": 1.0068621635437012, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9707998856902122, "clip_ratio/low_mean": 0.0301974443718791, "clip_ratio/low_min": 0.0301974443718791, "clip_ratio/high_mean": 0.09268568316474557, "clip_ratio/high_max": 0.09268568316474557, "clip_ratio/region_mean": 0.12288312753662467, "reward_total_mean": 0.7842920422554016, "reward_meter_mean": 0.7624353170394897, "reward_meter_std": 0.41185691952705383, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9511844515800476, "reward_repeat_soft_std": 0.039821553975343704, "reward_judge_quality_mean": 0.8737500309944153, "reward_judge_quality_std": 0.08601287752389908, "reward_total_composite_mean": 0.7842920422554016, "reward_total_composite_std": 0.24220286309719086} {"timestamp_utc": "2026-04-13T10:02:54Z", "mode": "train", "global_step": 1081, "epoch": 0.10858864892014063, "loss": 0.1389, "grad_norm": 10.372480392456055, "learning_rate": 6.7272727272727275e-06, "num_tokens": 1910092.0, "completions/mean_length": 80.875, "completions/min_length": 62.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.875, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.767886757850647, "rewards/meter/std": 0.24709437787532806, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9749447107315063, "rewards/repeat_soft/std": 0.014261425472795963, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.47303062677383423, "rewards/total_composite/std": 0.06528296321630478, "reward": 0.47303062677383423, "reward_std": 0.06528296321630478, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1676093339920044, "sampling/sampling_logp_difference/max": 2.04561710357666, "sampling/importance_sampling_ratio/min": 0.12930037081241608, "sampling/importance_sampling_ratio/mean": 1.0062720775604248, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4282553941011429, "clip_ratio/low_mean": 0.06684344448149204, "clip_ratio/low_min": 0.06684344448149204, "clip_ratio/high_mean": 0.07917457446455956, "clip_ratio/high_max": 0.07917457446455956, "clip_ratio/region_mean": 0.1460180189460516, "reward_total_mean": 0.47303062677383423, "reward_meter_mean": 0.767886757850647, "reward_meter_std": 0.24709437787532806, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9749447107315063, "reward_repeat_soft_std": 0.014261425472795963, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.47303062677383423, "reward_total_composite_std": 0.06528296321630478} {"timestamp_utc": "2026-04-13T10:03:00Z", "mode": "train", "global_step": 1082, "epoch": 0.10868910095429432, "loss": 0.0677, "grad_norm": 16.086347579956055, "learning_rate": 6.724242424242424e-06, "num_tokens": 1911612.0, "completions/mean_length": 42.0, "completions/min_length": 35.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.5908030271530151, "rewards/meter/std": 0.391478568315506, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9693589806556702, "rewards/repeat_soft/std": 0.029551459476351738, "rewards/judge_quality/mean": 0.6137499809265137, "rewards/judge_quality/std": 0.1927943229675293, "rewards/total_composite/mean": 0.5864404439926147, "rewards/total_composite/std": 0.1658206284046173, "reward": 0.5864404439926147, "reward_std": 0.1658206284046173, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16322758793830872, "sampling/sampling_logp_difference/max": 1.5092816352844238, "sampling/importance_sampling_ratio/min": 0.22106872498989105, "sampling/importance_sampling_ratio/mean": 1.0038870573043823, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0525820776820183, "clip_ratio/low_mean": 0.059773560613393784, "clip_ratio/low_min": 0.059773560613393784, "clip_ratio/high_mean": 0.08948134444653988, "clip_ratio/high_max": 0.08948134444653988, "clip_ratio/region_mean": 0.14925490505993366, "reward_total_mean": 0.5864404439926147, "reward_meter_mean": 0.5908030271530151, "reward_meter_std": 0.391478568315506, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9693589806556702, "reward_repeat_soft_std": 0.029551459476351738, "reward_judge_quality_mean": 0.6137499809265137, "reward_judge_quality_std": 0.1927943229675293, "reward_total_composite_mean": 0.5864404439926147, "reward_total_composite_std": 0.1658206284046173} {"timestamp_utc": "2026-04-13T10:03:07Z", "mode": "train", "global_step": 1083, "epoch": 0.10878955298844802, "loss": 0.0562, "grad_norm": 9.393478393554688, "learning_rate": 6.721212121212122e-06, "num_tokens": 1913940.0, "completions/mean_length": 89.0, "completions/min_length": 79.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.0, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.9875081777572632, "rewards/meter/std": 0.012425047345459461, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9484192132949829, "rewards/repeat_soft/std": 0.041187919676303864, "rewards/judge_quality/mean": 0.5325000286102295, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.6330901980400085, "rewards/total_composite/std": 0.09796515107154846, "reward": 0.6330901980400085, "reward_std": 0.09796513617038727, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14874045550823212, "sampling/sampling_logp_difference/max": 2.1184170246124268, "sampling/importance_sampling_ratio/min": 0.12022178620100021, "sampling/importance_sampling_ratio/mean": 1.0069530010223389, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.046143539249897, "clip_ratio/low_mean": 0.07416177168488503, "clip_ratio/low_min": 0.07416177168488503, "clip_ratio/high_mean": 0.06384505704045296, "clip_ratio/high_max": 0.06384505704045296, "clip_ratio/region_mean": 0.13800682872533798, "reward_total_mean": 0.6330901980400085, "reward_meter_mean": 0.9875081777572632, "reward_meter_std": 0.012425047345459461, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9484192132949829, "reward_repeat_soft_std": 0.041187919676303864, "reward_judge_quality_mean": 0.5325000286102295, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.6330901980400085, "reward_total_composite_std": 0.09796515107154846} {"timestamp_utc": "2026-04-13T10:03:14Z", "mode": "train", "global_step": 1084, "epoch": 0.10889000502260171, "loss": 0.024, "grad_norm": 9.318251609802246, "learning_rate": 6.718181818181819e-06, "num_tokens": 1916260.0, "completions/mean_length": 117.0, "completions/min_length": 99.0, "completions/max_length": 135.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 117.0, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 135.0, "rewards/meter/mean": 0.5858095288276672, "rewards/meter/std": 0.2507801949977875, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9901808500289917, "rewards/repeat_soft/std": 0.003913509659469128, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.43121761083602905, "rewards/total_composite/std": 0.1951574683189392, "reward": 0.43121761083602905, "reward_std": 0.1951574683189392, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16685812175273895, "sampling/sampling_logp_difference/max": 1.9139630794525146, "sampling/importance_sampling_ratio/min": 0.14749470353126526, "sampling/importance_sampling_ratio/mean": 1.0210882425308228, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4054247438907623, "clip_ratio/low_mean": 0.04495614022016525, "clip_ratio/low_min": 0.04495614022016525, "clip_ratio/high_mean": 0.1365183051675558, "clip_ratio/high_max": 0.1365183051675558, "clip_ratio/region_mean": 0.18147444538772106, "reward_total_mean": 0.43121761083602905, "reward_meter_mean": 0.5858095288276672, "reward_meter_std": 0.2507801949977875, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9901808500289917, "reward_repeat_soft_std": 0.003913509659469128, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.43121761083602905, "reward_total_composite_std": 0.1951574683189392} {"timestamp_utc": "2026-04-13T10:03:21Z", "mode": "train", "global_step": 1085, "epoch": 0.1089904570567554, "loss": -0.0325, "grad_norm": 13.496150016784668, "learning_rate": 6.715151515151516e-06, "num_tokens": 1918167.0, "completions/mean_length": 74.375, "completions/min_length": 63.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.375, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.6591826677322388, "rewards/meter/std": 0.3186195194721222, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9551384449005127, "rewards/repeat_soft/std": 0.022562414407730103, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.13695022463798523, "rewards/total_composite/mean": 0.546758770942688, "rewards/total_composite/std": 0.09543414413928986, "reward": 0.546758770942688, "reward_std": 0.09543413668870926, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14800848066806793, "sampling/sampling_logp_difference/max": 1.3822741508483887, "sampling/importance_sampling_ratio/min": 0.251007080078125, "sampling/importance_sampling_ratio/mean": 1.0303698778152466, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.157143548130989, "clip_ratio/low_mean": 0.04560697544366121, "clip_ratio/low_min": 0.04560697544366121, "clip_ratio/high_mean": 0.09135039150714874, "clip_ratio/high_max": 0.09135039150714874, "clip_ratio/region_mean": 0.13695736695080996, "reward_total_mean": 0.546758770942688, "reward_meter_mean": 0.6591826677322388, "reward_meter_std": 0.3186195194721222, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9551384449005127, "reward_repeat_soft_std": 0.022562414407730103, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.13695022463798523, "reward_total_composite_mean": 0.546758770942688, "reward_total_composite_std": 0.09543414413928986} {"timestamp_utc": "2026-04-13T10:03:28Z", "mode": "train", "global_step": 1086, "epoch": 0.10909090909090909, "loss": 0.14, "grad_norm": 14.193161964416504, "learning_rate": 6.712121212121213e-06, "num_tokens": 1920085.0, "completions/mean_length": 67.75, "completions/min_length": 51.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.75, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.8095401525497437, "rewards/meter/std": 0.2770191431045532, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9794746041297913, "rewards/repeat_soft/std": 0.012927144765853882, "rewards/judge_quality/mean": 0.4362499713897705, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.5207315683364868, "rewards/total_composite/std": 0.07252418994903564, "reward": 0.5207315683364868, "reward_std": 0.07252418249845505, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18057994544506073, "sampling/sampling_logp_difference/max": 1.7014508247375488, "sampling/importance_sampling_ratio/min": 0.18241867423057556, "sampling/importance_sampling_ratio/mean": 1.0027400255203247, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.337526798248291, "clip_ratio/low_mean": 0.05472476966679096, "clip_ratio/low_min": 0.05472476966679096, "clip_ratio/high_mean": 0.11002439074218273, "clip_ratio/high_max": 0.11002439074218273, "clip_ratio/region_mean": 0.1647491604089737, "reward_total_mean": 0.5207315683364868, "reward_meter_mean": 0.8095401525497437, "reward_meter_std": 0.2770191431045532, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9794746041297913, "reward_repeat_soft_std": 0.012927144765853882, "reward_judge_quality_mean": 0.4362499713897705, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.5207315683364868, "reward_total_composite_std": 0.07252418994903564} {"timestamp_utc": "2026-04-13T10:03:34Z", "mode": "train", "global_step": 1087, "epoch": 0.10919136112506278, "loss": 0.02, "grad_norm": 24.208295822143555, "learning_rate": 6.709090909090909e-06, "num_tokens": 1921455.0, "completions/mean_length": 22.25, "completions/min_length": 19.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.25, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.9190458059310913, "rewards/meter/std": 0.11271469295024872, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8771973252296448, "rewards/repeat_soft/std": 0.10158171504735947, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.12947696447372437, "rewards/total_composite/mean": 0.5464772582054138, "rewards/total_composite/std": 0.10019097477197647, "reward": 0.5464772582054138, "reward_std": 0.10019097477197647, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15317396819591522, "sampling/sampling_logp_difference/max": 0.9672427177429199, "sampling/importance_sampling_ratio/min": 0.38012972474098206, "sampling/importance_sampling_ratio/mean": 0.9914935827255249, "sampling/importance_sampling_ratio/max": 1.8522214889526367, "entropy": 1.1571223884820938, "clip_ratio/low_mean": 0.03112400509417057, "clip_ratio/low_min": 0.03112400509417057, "clip_ratio/high_mean": 0.09717909200116992, "clip_ratio/high_max": 0.09717909200116992, "clip_ratio/region_mean": 0.1283030970953405, "reward_total_mean": 0.5464772582054138, "reward_meter_mean": 0.9190458059310913, "reward_meter_std": 0.11271469295024872, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8771973252296448, "reward_repeat_soft_std": 0.10158171504735947, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.12947696447372437, "reward_total_composite_mean": 0.5464772582054138, "reward_total_composite_std": 0.10019097477197647} {"timestamp_utc": "2026-04-13T10:03:41Z", "mode": "train", "global_step": 1088, "epoch": 0.10929181315921647, "loss": 0.0118, "grad_norm": 9.507678985595703, "learning_rate": 6.706060606060607e-06, "num_tokens": 1923674.0, "completions/mean_length": 99.375, "completions/min_length": 91.0, "completions/max_length": 115.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 99.375, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 115.0, "rewards/meter/mean": 0.8243648409843445, "rewards/meter/std": 0.29020729660987854, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9430813789367676, "rewards/repeat_soft/std": 0.056274957954883575, "rewards/judge_quality/mean": 0.5074999928474426, "rewards/judge_quality/std": 0.1642080694437027, "rewards/total_composite/mean": 0.5826771855354309, "rewards/total_composite/std": 0.15603899955749512, "reward": 0.5826771855354309, "reward_std": 0.15603899955749512, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14087340235710144, "sampling/sampling_logp_difference/max": 1.4260168075561523, "sampling/importance_sampling_ratio/min": 0.24026404321193695, "sampling/importance_sampling_ratio/mean": 1.0190045833587646, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9441261067986488, "clip_ratio/low_mean": 0.09247690811753273, "clip_ratio/low_min": 0.09247690811753273, "clip_ratio/high_mean": 0.03256148658692837, "clip_ratio/high_max": 0.03256148658692837, "clip_ratio/region_mean": 0.1250383947044611, "reward_total_mean": 0.5826771855354309, "reward_meter_mean": 0.8243648409843445, "reward_meter_std": 0.29020729660987854, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9430813789367676, "reward_repeat_soft_std": 0.056274957954883575, "reward_judge_quality_mean": 0.5074999928474426, "reward_judge_quality_std": 0.1642080694437027, "reward_total_composite_mean": 0.5826771855354309, "reward_total_composite_std": 0.15603899955749512} {"timestamp_utc": "2026-04-13T10:03:47Z", "mode": "train", "global_step": 1089, "epoch": 0.10939226519337017, "loss": 0.0741, "grad_norm": 11.498926162719727, "learning_rate": 6.703030303030304e-06, "num_tokens": 1925438.0, "completions/mean_length": 48.5, "completions/min_length": 39.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.5, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.551418125629425, "rewards/meter/std": 0.3613598048686981, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9469898343086243, "rewards/repeat_soft/std": 0.033088844269514084, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.5358335375785828, "rewards/total_composite/std": 0.17223727703094482, "reward": 0.5358335375785828, "reward_std": 0.17223727703094482, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1180436760187149, "sampling/sampling_logp_difference/max": 1.4316868782043457, "sampling/importance_sampling_ratio/min": 0.23890559375286102, "sampling/importance_sampling_ratio/mean": 1.0067400932312012, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6848305761814117, "clip_ratio/low_mean": 0.07360310107469559, "clip_ratio/low_min": 0.07360310107469559, "clip_ratio/high_mean": 0.04411872383207083, "clip_ratio/high_max": 0.04411872383207083, "clip_ratio/region_mean": 0.11772182490676641, "reward_total_mean": 0.5358335375785828, "reward_meter_mean": 0.551418125629425, "reward_meter_std": 0.3613598048686981, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9469898343086243, "reward_repeat_soft_std": 0.033088844269514084, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.5358335375785828, "reward_total_composite_std": 0.17223727703094482} {"timestamp_utc": "2026-04-13T10:03:53Z", "mode": "train", "global_step": 1090, "epoch": 0.10949271722752386, "loss": 0.0789, "grad_norm": 23.13025665283203, "learning_rate": 6.700000000000001e-06, "num_tokens": 1926922.0, "completions/mean_length": 26.5, "completions/min_length": 24.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.5, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.9310585260391235, "rewards/meter/std": 0.08636492490768433, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9370556473731995, "rewards/repeat_soft/std": 0.057211827486753464, "rewards/judge_quality/mean": 0.6525000333786011, "rewards/judge_quality/std": 0.2921227812767029, "rewards/total_composite/mean": 0.7428244352340698, "rewards/total_composite/std": 0.20079685747623444, "reward": 0.7428244352340698, "reward_std": 0.20079685747623444, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17366978526115417, "sampling/sampling_logp_difference/max": 1.4532572031021118, "sampling/importance_sampling_ratio/min": 0.2338074892759323, "sampling/importance_sampling_ratio/mean": 1.015945315361023, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9537479653954506, "clip_ratio/low_mean": 0.06787037011235952, "clip_ratio/low_min": 0.06787037011235952, "clip_ratio/high_mean": 0.054067461751401424, "clip_ratio/high_max": 0.054067461751401424, "clip_ratio/region_mean": 0.12193783186376095, "reward_total_mean": 0.7428244352340698, "reward_meter_mean": 0.9310585260391235, "reward_meter_std": 0.08636492490768433, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9370556473731995, "reward_repeat_soft_std": 0.057211827486753464, "reward_judge_quality_mean": 0.6525000333786011, "reward_judge_quality_std": 0.2921227812767029, "reward_total_composite_mean": 0.7428244352340698, "reward_total_composite_std": 0.20079685747623444} {"timestamp_utc": "2026-04-13T10:03:59Z", "mode": "train", "global_step": 1091, "epoch": 0.10959316926167754, "loss": -0.0127, "grad_norm": 20.685056686401367, "learning_rate": 6.6969696969696975e-06, "num_tokens": 1928540.0, "completions/mean_length": 29.25, "completions/min_length": 26.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.25, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.5196160078048706, "rewards/meter/std": 0.4102964997291565, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.978695809841156, "rewards/repeat_soft/std": 0.02221783623099327, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.11563489586114883, "rewards/total_composite/mean": 0.5142635107040405, "rewards/total_composite/std": 0.14864948391914368, "reward": 0.5142635107040405, "reward_std": 0.14864948391914368, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16559728980064392, "sampling/sampling_logp_difference/max": 1.7340381145477295, "sampling/importance_sampling_ratio/min": 0.17656995356082916, "sampling/importance_sampling_ratio/mean": 0.9995377063751221, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.006347917020321, "clip_ratio/low_mean": 0.08393357694149017, "clip_ratio/low_min": 0.08393357694149017, "clip_ratio/high_mean": 0.062229438684880733, "clip_ratio/high_max": 0.062229438684880733, "clip_ratio/region_mean": 0.1461630156263709, "reward_total_mean": 0.5142635107040405, "reward_meter_mean": 0.5196160078048706, "reward_meter_std": 0.4102964997291565, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.978695809841156, "reward_repeat_soft_std": 0.02221783623099327, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.11563489586114883, "reward_total_composite_mean": 0.5142635107040405, "reward_total_composite_std": 0.14864948391914368} {"timestamp_utc": "2026-04-13T10:04:05Z", "mode": "train", "global_step": 1092, "epoch": 0.10969362129583124, "loss": 0.0369, "grad_norm": 11.980788230895996, "learning_rate": 6.693939393939395e-06, "num_tokens": 1930382.0, "completions/mean_length": 58.25, "completions/min_length": 51.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.25, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.6034842729568481, "rewards/meter/std": 0.3847072124481201, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9206419587135315, "rewards/repeat_soft/std": 0.0525461807847023, "rewards/judge_quality/mean": 0.4024999737739563, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.4950702488422394, "rewards/total_composite/std": 0.1105421930551529, "reward": 0.4950702488422394, "reward_std": 0.1105421930551529, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14994347095489502, "sampling/sampling_logp_difference/max": 1.428484559059143, "sampling/importance_sampling_ratio/min": 0.23967187106609344, "sampling/importance_sampling_ratio/mean": 1.0109158754348755, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9124277085065842, "clip_ratio/low_mean": 0.06228564865887165, "clip_ratio/low_min": 0.06228564865887165, "clip_ratio/high_mean": 0.06121163163334131, "clip_ratio/high_max": 0.06121163163334131, "clip_ratio/region_mean": 0.12349728029221296, "reward_total_mean": 0.4950702488422394, "reward_meter_mean": 0.6034842729568481, "reward_meter_std": 0.3847072124481201, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9206419587135315, "reward_repeat_soft_std": 0.0525461807847023, "reward_judge_quality_mean": 0.4024999737739563, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.4950702488422394, "reward_total_composite_std": 0.1105421930551529} {"timestamp_utc": "2026-04-13T10:04:12Z", "mode": "train", "global_step": 1093, "epoch": 0.10979407332998493, "loss": 0.0238, "grad_norm": 11.41279125213623, "learning_rate": 6.690909090909091e-06, "num_tokens": 1932357.0, "completions/mean_length": 70.875, "completions/min_length": 54.0, "completions/max_length": 97.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.875, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.9844741821289062, "rewards/meter/std": 0.018437247723340988, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9510754346847534, "rewards/repeat_soft/std": 0.021513454616069794, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5364227294921875, "rewards/total_composite/std": 0.010164810344576836, "reward": 0.5364227294921875, "reward_std": 0.010164814069867134, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14469408988952637, "sampling/sampling_logp_difference/max": 3.0628788471221924, "sampling/importance_sampling_ratio/min": 0.04675290733575821, "sampling/importance_sampling_ratio/mean": 1.026376724243164, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9509404972195625, "clip_ratio/low_mean": 0.09619823005050421, "clip_ratio/low_min": 0.09619823005050421, "clip_ratio/high_mean": 0.044693732634186745, "clip_ratio/high_max": 0.044693732634186745, "clip_ratio/region_mean": 0.14089196268469095, "reward_total_mean": 0.5364227294921875, "reward_meter_mean": 0.9844741821289062, "reward_meter_std": 0.018437247723340988, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9510754346847534, "reward_repeat_soft_std": 0.021513454616069794, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5364227294921875, "reward_total_composite_std": 0.010164810344576836} {"timestamp_utc": "2026-04-13T10:04:23Z", "mode": "train", "global_step": 1094, "epoch": 0.10989452536413863, "loss": -0.0871, "grad_norm": 2.331590175628662, "learning_rate": 6.687878787878788e-06, "num_tokens": 1933729.0, "completions/mean_length": 92.5, "completions/min_length": 30.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 32.57143020629883, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.6815160512924194, "rewards/meter/std": 0.2812528610229492, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9843109250068665, "rewards/repeat_soft/std": 0.011462407186627388, "rewards/judge_quality/mean": 0.8112499713897705, "rewards/judge_quality/std": 0.3075914680957794, "rewards/total_composite/mean": 0.6813645362854004, "rewards/total_composite/std": 0.31527772545814514, "reward": 0.6813645362854004, "reward_std": 0.31527772545814514, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12275287508964539, "sampling/sampling_logp_difference/max": 1.4223895072937012, "sampling/importance_sampling_ratio/min": 0.24113713204860687, "sampling/importance_sampling_ratio/mean": 1.011479139328003, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5784760490059853, "clip_ratio/low_mean": 0.011029412038624287, "clip_ratio/low_min": 0.011029412038624287, "clip_ratio/high_mean": 0.09784006420522928, "clip_ratio/high_max": 0.09784006420522928, "clip_ratio/region_mean": 0.10886947624385357, "reward_total_mean": 0.6813645362854004, "reward_meter_mean": 0.6815160512924194, "reward_meter_std": 0.2812528610229492, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9843109250068665, "reward_repeat_soft_std": 0.011462407186627388, "reward_judge_quality_mean": 0.8112499713897705, "reward_judge_quality_std": 0.3075914680957794, "reward_total_composite_mean": 0.6813645362854004, "reward_total_composite_std": 0.31527772545814514} {"timestamp_utc": "2026-04-13T10:04:30Z", "mode": "train", "global_step": 1095, "epoch": 0.10999497739829231, "loss": 0.0162, "grad_norm": 14.497970581054688, "learning_rate": 6.684848484848485e-06, "num_tokens": 1935420.0, "completions/mean_length": 49.375, "completions/min_length": 47.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.375, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.6142832040786743, "rewards/meter/std": 0.3878927230834961, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9937841892242432, "rewards/repeat_soft/std": 0.009870151989161968, "rewards/judge_quality/mean": 0.4724999964237213, "rewards/judge_quality/std": 0.19255799055099487, "rewards/total_composite/mean": 0.4929462671279907, "rewards/total_composite/std": 0.25837865471839905, "reward": 0.4929462671279907, "reward_std": 0.25837868452072144, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18936829268932343, "sampling/sampling_logp_difference/max": 2.8573012351989746, "sampling/importance_sampling_ratio/min": 0.2412366271018982, "sampling/importance_sampling_ratio/mean": 1.0276849269866943, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5982573479413986, "clip_ratio/low_mean": 0.07016889192163944, "clip_ratio/low_min": 0.07016889192163944, "clip_ratio/high_mean": 0.1261708401143551, "clip_ratio/high_max": 0.1261708401143551, "clip_ratio/region_mean": 0.19633973203599453, "reward_total_mean": 0.4929462671279907, "reward_meter_mean": 0.6142832040786743, "reward_meter_std": 0.3878927230834961, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9937841892242432, "reward_repeat_soft_std": 0.009870151989161968, "reward_judge_quality_mean": 0.4724999964237213, "reward_judge_quality_std": 0.19255799055099487, "reward_total_composite_mean": 0.4929462671279907, "reward_total_composite_std": 0.25837865471839905} {"timestamp_utc": "2026-04-13T10:04:36Z", "mode": "train", "global_step": 1096, "epoch": 0.110095429432446, "loss": -0.0192, "grad_norm": 11.968867301940918, "learning_rate": 6.681818181818183e-06, "num_tokens": 1937240.0, "completions/mean_length": 57.5, "completions/min_length": 53.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.5, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.7465670704841614, "rewards/meter/std": 0.24056415259838104, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9406276941299438, "rewards/repeat_soft/std": 0.035218942910432816, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.516263484954834, "rewards/total_composite/std": 0.05505356192588806, "reward": 0.516263484954834, "reward_std": 0.05505356192588806, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1491585373878479, "sampling/sampling_logp_difference/max": 1.6534686088562012, "sampling/importance_sampling_ratio/min": 0.1913849115371704, "sampling/importance_sampling_ratio/mean": 1.0211995840072632, "sampling/importance_sampling_ratio/max": 1.9647976160049438, "entropy": 1.1102144196629524, "clip_ratio/low_mean": 0.040419649332761765, "clip_ratio/low_min": 0.040419649332761765, "clip_ratio/high_mean": 0.08970674313604832, "clip_ratio/high_max": 0.08970674313604832, "clip_ratio/region_mean": 0.13012639246881008, "reward_total_mean": 0.516263484954834, "reward_meter_mean": 0.7465670704841614, "reward_meter_std": 0.24056415259838104, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9406276941299438, "reward_repeat_soft_std": 0.035218942910432816, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.516263484954834, "reward_total_composite_std": 0.05505356192588806} {"timestamp_utc": "2026-04-13T10:04:47Z", "mode": "train", "global_step": 1097, "epoch": 0.1101958814665997, "loss": -0.0776, "grad_norm": 1.4841713905334473, "learning_rate": 6.678787878787879e-06, "num_tokens": 1938710.0, "completions/mean_length": 86.75, "completions/min_length": 20.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 26.000001907348633, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.8709642887115479, "rewards/meter/std": 0.35193586349487305, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9601908326148987, "rewards/repeat_soft/std": 0.04054465889930725, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.13593590259552002, "rewards/total_composite/mean": 0.5243704319000244, "rewards/total_composite/std": 0.21498365700244904, "reward": 0.5243704319000244, "reward_std": 0.21498365700244904, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14930590987205505, "sampling/sampling_logp_difference/max": 1.9566144943237305, "sampling/importance_sampling_ratio/min": 0.1413360983133316, "sampling/importance_sampling_ratio/mean": 1.0057408809661865, "sampling/importance_sampling_ratio/max": 1.6734319925308228, "entropy": 0.8684336245059967, "clip_ratio/low_mean": 0.014880952425301075, "clip_ratio/low_min": 0.014880952425301075, "clip_ratio/high_mean": 0.06891872081905603, "clip_ratio/high_max": 0.06891872081905603, "clip_ratio/region_mean": 0.08379967324435711, "reward_total_mean": 0.5243704319000244, "reward_meter_mean": 0.8709642887115479, "reward_meter_std": 0.35193586349487305, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9601908326148987, "reward_repeat_soft_std": 0.04054465889930725, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.13593590259552002, "reward_total_composite_mean": 0.5243704319000244, "reward_total_composite_std": 0.21498365700244904} {"timestamp_utc": "2026-04-13T10:04:53Z", "mode": "train", "global_step": 1098, "epoch": 0.1102963335007534, "loss": -0.0076, "grad_norm": 13.859899520874023, "learning_rate": 6.6757575757575766e-06, "num_tokens": 1940332.0, "completions/mean_length": 43.75, "completions/min_length": 36.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.75, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.8528269529342651, "rewards/meter/std": 0.23338685929775238, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9464430809020996, "rewards/repeat_soft/std": 0.03395868465304375, "rewards/judge_quality/mean": 0.5012500286102295, "rewards/judge_quality/std": 0.16974246501922607, "rewards/total_composite/mean": 0.6254202127456665, "rewards/total_composite/std": 0.13749991357326508, "reward": 0.6254202127456665, "reward_std": 0.13749991357326508, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15951161086559296, "sampling/sampling_logp_difference/max": 2.5978305339813232, "sampling/importance_sampling_ratio/min": 0.07443489134311676, "sampling/importance_sampling_ratio/mean": 1.0151758193969727, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.996156208217144, "clip_ratio/low_mean": 0.09627203363925219, "clip_ratio/low_min": 0.09627203363925219, "clip_ratio/high_mean": 0.043026004917919636, "clip_ratio/high_max": 0.043026004917919636, "clip_ratio/region_mean": 0.13929803855717182, "reward_total_mean": 0.6254202127456665, "reward_meter_mean": 0.8528269529342651, "reward_meter_std": 0.23338685929775238, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9464430809020996, "reward_repeat_soft_std": 0.03395868465304375, "reward_judge_quality_mean": 0.5012500286102295, "reward_judge_quality_std": 0.16974246501922607, "reward_total_composite_mean": 0.6254202127456665, "reward_total_composite_std": 0.13749991357326508} {"timestamp_utc": "2026-04-13T10:05:00Z", "mode": "train", "global_step": 1099, "epoch": 0.11039678553490709, "loss": 0.0229, "grad_norm": 11.417180061340332, "learning_rate": 6.672727272727273e-06, "num_tokens": 1942676.0, "completions/mean_length": 96.0, "completions/min_length": 76.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.0, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.4464265704154968, "rewards/meter/std": 0.36981698870658875, "rewards/count_adherence/mean": 0.7749999761581421, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9601141214370728, "rewards/repeat_soft/std": 0.04743463173508644, "rewards/judge_quality/mean": 0.5024999976158142, "rewards/judge_quality/std": 0.1348809152841568, "rewards/total_composite/mean": 0.38410118222236633, "rewards/total_composite/std": 0.19046717882156372, "reward": 0.38410118222236633, "reward_std": 0.19046717882156372, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14962473511695862, "sampling/sampling_logp_difference/max": 2.2815001010894775, "sampling/importance_sampling_ratio/min": 0.10213088244199753, "sampling/importance_sampling_ratio/mean": 0.9960708022117615, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9751346260309219, "clip_ratio/low_mean": 0.0620602834969759, "clip_ratio/low_min": 0.0620602834969759, "clip_ratio/high_mean": 0.07958950288593769, "clip_ratio/high_max": 0.07958950288593769, "clip_ratio/region_mean": 0.1416497863829136, "reward_total_mean": 0.38410118222236633, "reward_meter_mean": 0.4464265704154968, "reward_meter_std": 0.36981698870658875, "reward_count_adherence_mean": 0.7749999761581421, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9601141214370728, "reward_repeat_soft_std": 0.04743463173508644, "reward_judge_quality_mean": 0.5024999976158142, "reward_judge_quality_std": 0.1348809152841568, "reward_total_composite_mean": 0.38410118222236633, "reward_total_composite_std": 0.19046717882156372} {"timestamp_utc": "2026-04-13T10:05:06Z", "mode": "train", "global_step": 1100, "epoch": 0.11049723756906077, "loss": 0.0269, "grad_norm": 14.642799377441406, "learning_rate": 6.66969696969697e-06, "num_tokens": 1944623.0, "completions/mean_length": 62.375, "completions/min_length": 58.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.375, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.7071465253829956, "rewards/meter/std": 0.30786022543907166, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9604854583740234, "rewards/repeat_soft/std": 0.026317918673157692, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.4587589502334595, "rewards/total_composite/std": 0.09426096081733704, "reward": 0.4587589502334595, "reward_std": 0.09426096826791763, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16210375726222992, "sampling/sampling_logp_difference/max": 1.7364449501037598, "sampling/importance_sampling_ratio/min": 0.1761454939842224, "sampling/importance_sampling_ratio/mean": 1.0446785688400269, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2134244069457054, "clip_ratio/low_mean": 0.09298294503241777, "clip_ratio/low_min": 0.09298294503241777, "clip_ratio/high_mean": 0.030166194774210453, "clip_ratio/high_max": 0.030166194774210453, "clip_ratio/region_mean": 0.12314913980662823, "reward_total_mean": 0.4587589502334595, "reward_meter_mean": 0.7071465253829956, "reward_meter_std": 0.30786022543907166, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9604854583740234, "reward_repeat_soft_std": 0.026317918673157692, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.4587589502334595, "reward_total_composite_std": 0.09426096081733704} {"timestamp_utc": "2026-04-13T10:05:45Z", "mode": "eval", "global_step": 1100, "epoch": 0.11049723756906077, "eval_loss": NaN, "eval_runtime": 38.0621, "eval_samples_per_second": 2.102, "eval_steps_per_second": 0.263, "eval_num_tokens": 1944623.0, "eval_completions/mean_length": 59.0875, "eval_completions/min_length": 30.1, "eval_completions/max_length": 87.9, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 59.0875, "eval_completions/min_terminated_length": 30.1, "eval_completions/max_terminated_length": 87.9, "eval_rewards/meter/mean": 0.7739002108573914, "eval_rewards/meter/std": 0.3174647897481918, "eval_rewards/count_adherence/mean": 0.8747916758060456, "eval_rewards/count_adherence/std": 0.13870810568332673, "eval_rewards/hard_gate/mean": 0.9625, "eval_rewards/hard_gate/std": 0.10606601536273956, "eval_rewards/repeat_soft/mean": 0.9302830159664154, "eval_rewards/repeat_soft/std": 0.05693140346556902, "eval_rewards/judge_quality/mean": 0.47762500047683715, "eval_rewards/judge_quality/std": 0.16730604767799379, "eval_rewards/total_composite/mean": 0.5368325889110566, "eval_rewards/total_composite/std": 0.1783295027911663, "eval_reward": 0.5368325889110566, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.08279976770281791, "eval_sampling/sampling_logp_difference/max": 1.016366195678711, "eval_sampling/importance_sampling_ratio/min": 0.3745600953698158, "eval_sampling/importance_sampling_ratio/mean": 1.0196762204170227, "eval_sampling/importance_sampling_ratio/max": 1.4289889454841613, "eval_entropy": 0.9940823078155517, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5368325889110566, "eval_reward_meter_mean": 0.7739002108573914, "eval_reward_meter_std": 0.3174647897481918, "eval_reward_count_adherence_mean": 0.8747916758060456, "eval_reward_count_adherence_std": 0.13870810568332673, "eval_reward_hard_gate_mean": 0.9625, "eval_reward_hard_gate_std": 0.10606601536273956, "eval_reward_repeat_soft_mean": 0.9302830159664154, "eval_reward_repeat_soft_std": 0.05693140346556902, "eval_reward_judge_quality_mean": 0.47762500047683715, "eval_reward_judge_quality_std": 0.16730604767799379, "eval_reward_total_composite_mean": 0.5368325889110566, "eval_reward_total_composite_std": 0.1783295027911663} {"timestamp_utc": "2026-04-13T10:05:54Z", "mode": "train", "global_step": 1101, "epoch": 0.11059768960321446, "loss": 0.0135, "grad_norm": 9.57143497467041, "learning_rate": 6.666666666666667e-06, "num_tokens": 1946331.0, "completions/mean_length": 51.5, "completions/min_length": 42.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.5, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9508030414581299, "rewards/meter/std": 0.09549621492624283, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9613487124443054, "rewards/repeat_soft/std": 0.039299752563238144, "rewards/judge_quality/mean": 0.5237500071525574, "rewards/judge_quality/std": 0.19078317284584045, "rewards/total_composite/mean": 0.6705681681632996, "rewards/total_composite/std": 0.13356658816337585, "reward": 0.6705681681632996, "reward_std": 0.13356660306453705, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12682457268238068, "sampling/sampling_logp_difference/max": 1.6577732563018799, "sampling/importance_sampling_ratio/min": 0.19056284427642822, "sampling/importance_sampling_ratio/mean": 1.0136646032333374, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9868056029081345, "clip_ratio/low_mean": 0.08438512543216348, "clip_ratio/low_min": 0.08438512543216348, "clip_ratio/high_mean": 0.05235273018479347, "clip_ratio/high_max": 0.05235273018479347, "clip_ratio/region_mean": 0.13673785561695695, "reward_total_mean": 0.6705681681632996, "reward_meter_mean": 0.9508030414581299, "reward_meter_std": 0.09549621492624283, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9613487124443054, "reward_repeat_soft_std": 0.039299752563238144, "reward_judge_quality_mean": 0.5237500071525574, "reward_judge_quality_std": 0.19078317284584045, "reward_total_composite_mean": 0.6705681681632996, "reward_total_composite_std": 0.13356658816337585} {"timestamp_utc": "2026-04-13T10:06:01Z", "mode": "train", "global_step": 1102, "epoch": 0.11069814163736816, "loss": 0.0406, "grad_norm": 9.237506866455078, "learning_rate": 6.663636363636365e-06, "num_tokens": 1948547.0, "completions/mean_length": 65.0, "completions/min_length": 61.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9411215782165527, "rewards/meter/std": 0.06898335367441177, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.910030722618103, "rewards/repeat_soft/std": 0.0485076941549778, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.5892764925956726, "rewards/total_composite/std": 0.09233304858207703, "reward": 0.5892764925956726, "reward_std": 0.09233304113149643, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12735266983509064, "sampling/sampling_logp_difference/max": 1.9520063400268555, "sampling/importance_sampling_ratio/min": 0.14198890328407288, "sampling/importance_sampling_ratio/mean": 0.9912409782409668, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6516686156392097, "clip_ratio/low_mean": 0.08202955033630133, "clip_ratio/low_min": 0.08202955033630133, "clip_ratio/high_mean": 0.0356006845831871, "clip_ratio/high_max": 0.0356006845831871, "clip_ratio/region_mean": 0.11763023491948843, "reward_total_mean": 0.5892764925956726, "reward_meter_mean": 0.9411215782165527, "reward_meter_std": 0.06898335367441177, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.910030722618103, "reward_repeat_soft_std": 0.0485076941549778, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.5892764925956726, "reward_total_composite_std": 0.09233304858207703} {"timestamp_utc": "2026-04-13T10:06:08Z", "mode": "train", "global_step": 1103, "epoch": 0.11079859367152185, "loss": 0.0492, "grad_norm": 8.690917015075684, "learning_rate": 6.660606060606061e-06, "num_tokens": 1950482.0, "completions/mean_length": 69.875, "completions/min_length": 51.0, "completions/max_length": 92.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.875, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 92.0, "rewards/meter/mean": 0.9910345077514648, "rewards/meter/std": 0.006791371386498213, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9173648357391357, "rewards/repeat_soft/std": 0.03406515344977379, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.5605899691581726, "rewards/total_composite/std": 0.010260870680212975, "reward": 0.5605899691581726, "reward_std": 0.010260866023600101, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14378008246421814, "sampling/sampling_logp_difference/max": 1.3936083316802979, "sampling/importance_sampling_ratio/min": 0.24817818403244019, "sampling/importance_sampling_ratio/mean": 1.0004949569702148, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1420028284192085, "clip_ratio/low_mean": 0.07078136503696442, "clip_ratio/low_min": 0.07078136503696442, "clip_ratio/high_mean": 0.049270967952907085, "clip_ratio/high_max": 0.049270967952907085, "clip_ratio/region_mean": 0.1200523329898715, "reward_total_mean": 0.5605899691581726, "reward_meter_mean": 0.9910345077514648, "reward_meter_std": 0.006791371386498213, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9173648357391357, "reward_repeat_soft_std": 0.03406515344977379, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.5605899691581726, "reward_total_composite_std": 0.010260870680212975} {"timestamp_utc": "2026-04-13T10:06:15Z", "mode": "train", "global_step": 1104, "epoch": 0.11089904570567553, "loss": -0.0089, "grad_norm": 12.141768455505371, "learning_rate": 6.657575757575758e-06, "num_tokens": 1952103.0, "completions/mean_length": 43.625, "completions/min_length": 37.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.625, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.47106683254241943, "rewards/meter/std": 0.40478745102882385, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9712631702423096, "rewards/repeat_soft/std": 0.017058251425623894, "rewards/judge_quality/mean": 0.6025000214576721, "rewards/judge_quality/std": 0.1976107507944107, "rewards/total_composite/mean": 0.517354428768158, "rewards/total_composite/std": 0.14876633882522583, "reward": 0.517354428768158, "reward_std": 0.14876633882522583, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14809371531009674, "sampling/sampling_logp_difference/max": 1.211209774017334, "sampling/importance_sampling_ratio/min": 0.2978367507457733, "sampling/importance_sampling_ratio/mean": 1.0351276397705078, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1666222214698792, "clip_ratio/low_mean": 0.09970238246023655, "clip_ratio/low_min": 0.09970238246023655, "clip_ratio/high_mean": 0.06661933101713657, "clip_ratio/high_max": 0.06661933101713657, "clip_ratio/region_mean": 0.16632171347737312, "reward_total_mean": 0.517354428768158, "reward_meter_mean": 0.47106683254241943, "reward_meter_std": 0.40478745102882385, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9712631702423096, "reward_repeat_soft_std": 0.017058251425623894, "reward_judge_quality_mean": 0.6025000214576721, "reward_judge_quality_std": 0.1976107507944107, "reward_total_composite_mean": 0.517354428768158, "reward_total_composite_std": 0.14876633882522583} {"timestamp_utc": "2026-04-13T10:06:22Z", "mode": "train", "global_step": 1105, "epoch": 0.11099949773982923, "loss": 0.0742, "grad_norm": 8.783102989196777, "learning_rate": 6.654545454545455e-06, "num_tokens": 1954012.0, "completions/mean_length": 86.625, "completions/min_length": 77.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.625, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.9817788600921631, "rewards/meter/std": 0.013821211643517017, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9379310607910156, "rewards/repeat_soft/std": 0.04740970954298973, "rewards/judge_quality/mean": 0.6074999570846558, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.6879007816314697, "rewards/total_composite/std": 0.09658631682395935, "reward": 0.6879007816314697, "reward_std": 0.09658631682395935, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1555800437927246, "sampling/sampling_logp_difference/max": 1.3972806930541992, "sampling/importance_sampling_ratio/min": 0.2472684532403946, "sampling/importance_sampling_ratio/mean": 1.0163252353668213, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1243841573596, "clip_ratio/low_mean": 0.05159039422869682, "clip_ratio/low_min": 0.05159039422869682, "clip_ratio/high_mean": 0.09754106309264898, "clip_ratio/high_max": 0.09754106309264898, "clip_ratio/region_mean": 0.1491314573213458, "reward_total_mean": 0.6879007816314697, "reward_meter_mean": 0.9817788600921631, "reward_meter_std": 0.013821211643517017, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9379310607910156, "reward_repeat_soft_std": 0.04740970954298973, "reward_judge_quality_mean": 0.6074999570846558, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.6879007816314697, "reward_total_composite_std": 0.09658631682395935} {"timestamp_utc": "2026-04-13T10:06:27Z", "mode": "train", "global_step": 1106, "epoch": 0.11109994977398292, "loss": 0.1157, "grad_norm": 20.9514217376709, "learning_rate": 6.651515151515152e-06, "num_tokens": 1955611.0, "completions/mean_length": 25.875, "completions/min_length": 20.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.875, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.9920638799667358, "rewards/meter/std": 0.006432940252125263, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9376035928726196, "rewards/repeat_soft/std": 0.06885240226984024, "rewards/judge_quality/mean": 0.33375000953674316, "rewards/judge_quality/std": 0.12117726355791092, "rewards/total_composite/mean": 0.5561680793762207, "rewards/total_composite/std": 0.08233477920293808, "reward": 0.5561680793762207, "reward_std": 0.08233474940061569, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1665658801794052, "sampling/sampling_logp_difference/max": 1.387542724609375, "sampling/importance_sampling_ratio/min": 0.24968811869621277, "sampling/importance_sampling_ratio/mean": 1.0338551998138428, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3538757637143135, "clip_ratio/low_mean": 0.10827161557972431, "clip_ratio/low_min": 0.10827161557972431, "clip_ratio/high_mean": 0.08063207007944584, "clip_ratio/high_max": 0.08063207007944584, "clip_ratio/region_mean": 0.18890368565917015, "reward_total_mean": 0.5561680793762207, "reward_meter_mean": 0.9920638799667358, "reward_meter_std": 0.006432940252125263, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9376035928726196, "reward_repeat_soft_std": 0.06885240226984024, "reward_judge_quality_mean": 0.33375000953674316, "reward_judge_quality_std": 0.12117726355791092, "reward_total_composite_mean": 0.5561680793762207, "reward_total_composite_std": 0.08233477920293808} {"timestamp_utc": "2026-04-13T10:06:33Z", "mode": "train", "global_step": 1107, "epoch": 0.11120040180813662, "loss": 0.0631, "grad_norm": 12.246903419494629, "learning_rate": 6.6484848484848485e-06, "num_tokens": 1957230.0, "completions/mean_length": 38.375, "completions/min_length": 33.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.375, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.9808336496353149, "rewards/meter/std": 0.0065889740362763405, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8884369134902954, "rewards/repeat_soft/std": 0.05048581212759018, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.6058225631713867, "rewards/total_composite/std": 0.0140786562114954, "reward": 0.6058225631713867, "reward_std": 0.014078662730753422, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12168633192777634, "sampling/sampling_logp_difference/max": 1.398416519165039, "sampling/importance_sampling_ratio/min": 0.24698776006698608, "sampling/importance_sampling_ratio/mean": 1.0060867071151733, "sampling/importance_sampling_ratio/max": 1.7908825874328613, "entropy": 0.9118454605340958, "clip_ratio/low_mean": 0.04358908720314503, "clip_ratio/low_min": 0.04358908720314503, "clip_ratio/high_mean": 0.06082251248881221, "clip_ratio/high_max": 0.06082251248881221, "clip_ratio/region_mean": 0.10441159969195724, "reward_total_mean": 0.6058225631713867, "reward_meter_mean": 0.9808336496353149, "reward_meter_std": 0.0065889740362763405, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8884369134902954, "reward_repeat_soft_std": 0.05048581212759018, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.6058225631713867, "reward_total_composite_std": 0.0140786562114954} {"timestamp_utc": "2026-04-13T10:06:39Z", "mode": "train", "global_step": 1108, "epoch": 0.11130085384229031, "loss": 0.0263, "grad_norm": 15.682332038879395, "learning_rate": 6.645454545454546e-06, "num_tokens": 1958885.0, "completions/mean_length": 43.875, "completions/min_length": 41.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.875, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.5969927310943604, "rewards/meter/std": 0.36882877349853516, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8888386487960815, "rewards/repeat_soft/std": 0.038945961743593216, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.5122449398040771, "rewards/total_composite/std": 0.11345456540584564, "reward": 0.5122449398040771, "reward_std": 0.11345456540584564, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14469872415065765, "sampling/sampling_logp_difference/max": 4.074991226196289, "sampling/importance_sampling_ratio/min": 0.016992364078760147, "sampling/importance_sampling_ratio/mean": 1.0096375942230225, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7018393352627754, "clip_ratio/low_mean": 0.019392292946577072, "clip_ratio/low_min": 0.019392292946577072, "clip_ratio/high_mean": 0.06644094968214631, "clip_ratio/high_max": 0.06644094968214631, "clip_ratio/region_mean": 0.08583324262872338, "reward_total_mean": 0.5122449398040771, "reward_meter_mean": 0.5969927310943604, "reward_meter_std": 0.36882877349853516, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8888386487960815, "reward_repeat_soft_std": 0.038945961743593216, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.5122449398040771, "reward_total_composite_std": 0.11345456540584564} {"timestamp_utc": "2026-04-13T10:06:46Z", "mode": "train", "global_step": 1109, "epoch": 0.11140130587644399, "loss": -0.0118, "grad_norm": 11.471404075622559, "learning_rate": 6.642424242424242e-06, "num_tokens": 1960766.0, "completions/mean_length": 56.125, "completions/min_length": 52.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.125, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.6876039505004883, "rewards/meter/std": 0.3078877031803131, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7964769601821899, "rewards/repeat_soft/std": 0.05928468331694603, "rewards/judge_quality/mean": 0.41499999165534973, "rewards/judge_quality/std": 0.14520922303199768, "rewards/total_composite/mean": 0.4661751687526703, "rewards/total_composite/std": 0.13284611701965332, "reward": 0.4661751687526703, "reward_std": 0.13284608721733093, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11511439085006714, "sampling/sampling_logp_difference/max": 1.5071403980255127, "sampling/importance_sampling_ratio/min": 0.2215426117181778, "sampling/importance_sampling_ratio/mean": 1.023069977760315, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8226094245910645, "clip_ratio/low_mean": 0.06764401262626052, "clip_ratio/low_min": 0.06764401262626052, "clip_ratio/high_mean": 0.039518462494015694, "clip_ratio/high_max": 0.039518462494015694, "clip_ratio/region_mean": 0.10716247512027621, "reward_total_mean": 0.4661751687526703, "reward_meter_mean": 0.6876039505004883, "reward_meter_std": 0.3078877031803131, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7964769601821899, "reward_repeat_soft_std": 0.05928468331694603, "reward_judge_quality_mean": 0.41499999165534973, "reward_judge_quality_std": 0.14520922303199768, "reward_total_composite_mean": 0.4661751687526703, "reward_total_composite_std": 0.13284611701965332} {"timestamp_utc": "2026-04-13T10:06:52Z", "mode": "train", "global_step": 1110, "epoch": 0.11150175791059769, "loss": 0.0428, "grad_norm": 9.730709075927734, "learning_rate": 6.63939393939394e-06, "num_tokens": 1962938.0, "completions/mean_length": 77.5, "completions/min_length": 64.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.5, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.9428097009658813, "rewards/meter/std": 0.12286613881587982, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8957542181015015, "rewards/repeat_soft/std": 0.037643659859895706, "rewards/judge_quality/mean": 0.4024999737739563, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.5306164026260376, "rewards/total_composite/std": 0.05158373340964317, "reward": 0.5306164026260376, "reward_std": 0.05158371478319168, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1261560469865799, "sampling/sampling_logp_difference/max": 1.6344184875488281, "sampling/importance_sampling_ratio/min": 0.2225971221923828, "sampling/importance_sampling_ratio/mean": 1.0119463205337524, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8850665241479874, "clip_ratio/low_mean": 0.012559335213154554, "clip_ratio/low_min": 0.012559335213154554, "clip_ratio/high_mean": 0.08973471214994788, "clip_ratio/high_max": 0.08973471214994788, "clip_ratio/region_mean": 0.10229404736310244, "reward_total_mean": 0.5306164026260376, "reward_meter_mean": 0.9428097009658813, "reward_meter_std": 0.12286613881587982, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8957542181015015, "reward_repeat_soft_std": 0.037643659859895706, "reward_judge_quality_mean": 0.4024999737739563, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.5306164026260376, "reward_total_composite_std": 0.05158373340964317} {"timestamp_utc": "2026-04-13T10:06:58Z", "mode": "train", "global_step": 1111, "epoch": 0.11160220994475138, "loss": 0.0351, "grad_norm": 12.884098052978516, "learning_rate": 6.6363636363636375e-06, "num_tokens": 1964605.0, "completions/mean_length": 47.375, "completions/min_length": 44.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.375, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.8017257452011108, "rewards/meter/std": 0.30836212635040283, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9168136119842529, "rewards/repeat_soft/std": 0.041258662939071655, "rewards/judge_quality/mean": 0.4437499940395355, "rewards/judge_quality/std": 0.128834068775177, "rewards/total_composite/mean": 0.5684049725532532, "rewards/total_composite/std": 0.11981135606765747, "reward": 0.5684049725532532, "reward_std": 0.11981134861707687, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11596706509590149, "sampling/sampling_logp_difference/max": 2.4830503463745117, "sampling/importance_sampling_ratio/min": 0.08348816633224487, "sampling/importance_sampling_ratio/mean": 1.022456169128418, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7652312889695168, "clip_ratio/low_mean": 0.055207240860909224, "clip_ratio/low_min": 0.055207240860909224, "clip_ratio/high_mean": 0.05440958496183157, "clip_ratio/high_max": 0.05440958496183157, "clip_ratio/region_mean": 0.1096168258227408, "reward_total_mean": 0.5684049725532532, "reward_meter_mean": 0.8017257452011108, "reward_meter_std": 0.30836212635040283, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9168136119842529, "reward_repeat_soft_std": 0.041258662939071655, "reward_judge_quality_mean": 0.4437499940395355, "reward_judge_quality_std": 0.128834068775177, "reward_total_composite_mean": 0.5684049725532532, "reward_total_composite_std": 0.11981135606765747} {"timestamp_utc": "2026-04-13T10:07:05Z", "mode": "train", "global_step": 1112, "epoch": 0.11170266197890508, "loss": -0.0934, "grad_norm": 12.25678825378418, "learning_rate": 6.633333333333334e-06, "num_tokens": 1966645.0, "completions/mean_length": 67.0, "completions/min_length": 53.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.9791103601455688, "rewards/meter/std": 0.010904289782047272, "rewards/count_adherence/mean": 0.7000000476837158, "rewards/count_adherence/std": 0.10690449178218842, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8605831861495972, "rewards/repeat_soft/std": 0.08659627288579941, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.5227519869804382, "rewards/total_composite/std": 0.049343012273311615, "reward": 0.5227519869804382, "reward_std": 0.049343008548021317, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15082481503486633, "sampling/sampling_logp_difference/max": 1.8279852867126465, "sampling/importance_sampling_ratio/min": 0.160737082362175, "sampling/importance_sampling_ratio/mean": 1.0180777311325073, "sampling/importance_sampling_ratio/max": 1.8894423246383667, "entropy": 1.0666228979825974, "clip_ratio/low_mean": 0.05456842668354511, "clip_ratio/low_min": 0.05456842668354511, "clip_ratio/high_mean": 0.09077512472867966, "clip_ratio/high_max": 0.09077512472867966, "clip_ratio/region_mean": 0.14534355141222477, "reward_total_mean": 0.5227519869804382, "reward_meter_mean": 0.9791103601455688, "reward_meter_std": 0.010904289782047272, "reward_count_adherence_mean": 0.7000000476837158, "reward_count_adherence_std": 0.10690449178218842, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8605831861495972, "reward_repeat_soft_std": 0.08659627288579941, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.5227519869804382, "reward_total_composite_std": 0.049343012273311615} {"timestamp_utc": "2026-04-13T10:07:16Z", "mode": "train", "global_step": 1113, "epoch": 0.11180311401305876, "loss": -0.1674, "grad_norm": 3.0614638328552246, "learning_rate": 6.630303030303031e-06, "num_tokens": 1968481.0, "completions/mean_length": 120.5, "completions/min_length": 54.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 64.5714340209961, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.8282565474510193, "rewards/meter/std": 0.32117608189582825, "rewards/count_adherence/mean": 0.71875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9734363555908203, "rewards/repeat_soft/std": 0.014392375946044922, "rewards/judge_quality/mean": 0.4112499952316284, "rewards/judge_quality/std": 0.1797965168952942, "rewards/total_composite/mean": 0.5041776895523071, "rewards/total_composite/std": 0.22154462337493896, "reward": 0.5041776895523071, "reward_std": 0.22154460847377777, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1511511206626892, "sampling/sampling_logp_difference/max": 1.5392913818359375, "sampling/importance_sampling_ratio/min": 0.21453307569026947, "sampling/importance_sampling_ratio/mean": 1.0373806953430176, "sampling/importance_sampling_ratio/max": 1.993828296661377, "entropy": 1.2479336559772491, "clip_ratio/low_mean": 0.025462962687015533, "clip_ratio/low_min": 0.025462962687015533, "clip_ratio/high_mean": 0.10990368854254484, "clip_ratio/high_max": 0.10990368854254484, "clip_ratio/region_mean": 0.13536665122956038, "reward_total_mean": 0.5041776895523071, "reward_meter_mean": 0.8282565474510193, "reward_meter_std": 0.32117608189582825, "reward_count_adherence_mean": 0.71875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9734363555908203, "reward_repeat_soft_std": 0.014392375946044922, "reward_judge_quality_mean": 0.4112499952316284, "reward_judge_quality_std": 0.1797965168952942, "reward_total_composite_mean": 0.5041776895523071, "reward_total_composite_std": 0.22154462337493896} {"timestamp_utc": "2026-04-13T10:07:22Z", "mode": "train", "global_step": 1114, "epoch": 0.11190356604721245, "loss": 0.0179, "grad_norm": 13.786949157714844, "learning_rate": 6.627272727272728e-06, "num_tokens": 1970614.0, "completions/mean_length": 69.625, "completions/min_length": 54.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.625, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.9920821189880371, "rewards/meter/std": 0.005062623415142298, "rewards/count_adherence/mean": 0.6000000238418579, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8936679363250732, "rewards/repeat_soft/std": 0.05867590755224228, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.5490313172340393, "rewards/total_composite/std": 0.07378941774368286, "reward": 0.5490313172340393, "reward_std": 0.07378942519426346, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1403108388185501, "sampling/sampling_logp_difference/max": 1.1537199020385742, "sampling/importance_sampling_ratio/min": 0.31546109914779663, "sampling/importance_sampling_ratio/mean": 1.0259828567504883, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0104362666606903, "clip_ratio/low_mean": 0.11913142912089825, "clip_ratio/low_min": 0.11913142912089825, "clip_ratio/high_mean": 0.00844594556838274, "clip_ratio/high_max": 0.00844594556838274, "clip_ratio/region_mean": 0.127577374689281, "reward_total_mean": 0.5490313172340393, "reward_meter_mean": 0.9920821189880371, "reward_meter_std": 0.005062623415142298, "reward_count_adherence_mean": 0.6000000238418579, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8936679363250732, "reward_repeat_soft_std": 0.05867590755224228, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.5490313172340393, "reward_total_composite_std": 0.07378941774368286} {"timestamp_utc": "2026-04-13T10:07:29Z", "mode": "train", "global_step": 1115, "epoch": 0.11200401808136615, "loss": 0.0496, "grad_norm": 10.43582534790039, "learning_rate": 6.624242424242425e-06, "num_tokens": 1972654.0, "completions/mean_length": 69.0, "completions/min_length": 58.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.0, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.6927752494812012, "rewards/meter/std": 0.350618451833725, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9682995080947876, "rewards/repeat_soft/std": 0.024794267490506172, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.10507651418447495, "rewards/total_composite/mean": 0.5085471868515015, "rewards/total_composite/std": 0.1273602545261383, "reward": 0.5085471868515015, "reward_std": 0.1273602545261383, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15537992119789124, "sampling/sampling_logp_difference/max": 1.9183073043823242, "sampling/importance_sampling_ratio/min": 0.14685533940792084, "sampling/importance_sampling_ratio/mean": 1.0243021249771118, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1463446840643883, "clip_ratio/low_mean": 0.0369216650724411, "clip_ratio/low_min": 0.0369216650724411, "clip_ratio/high_mean": 0.08321494609117508, "clip_ratio/high_max": 0.08321494609117508, "clip_ratio/region_mean": 0.12013661116361618, "reward_total_mean": 0.5085471868515015, "reward_meter_mean": 0.6927752494812012, "reward_meter_std": 0.350618451833725, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9682995080947876, "reward_repeat_soft_std": 0.024794267490506172, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.10507651418447495, "reward_total_composite_mean": 0.5085471868515015, "reward_total_composite_std": 0.1273602545261383} {"timestamp_utc": "2026-04-13T10:07:35Z", "mode": "train", "global_step": 1116, "epoch": 0.11210447011551984, "loss": 0.0279, "grad_norm": 12.833725929260254, "learning_rate": 6.621212121212121e-06, "num_tokens": 1974263.0, "completions/mean_length": 43.125, "completions/min_length": 38.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.125, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.7856926918029785, "rewards/meter/std": 0.21308495104312897, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9201638102531433, "rewards/repeat_soft/std": 0.08705054223537445, "rewards/judge_quality/mean": 0.5487500429153442, "rewards/judge_quality/std": 0.22937415540218353, "rewards/total_composite/mean": 0.6049907207489014, "rewards/total_composite/std": 0.10036706924438477, "reward": 0.6049907207489014, "reward_std": 0.10036708414554596, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14687471091747284, "sampling/sampling_logp_difference/max": 2.573352336883545, "sampling/importance_sampling_ratio/min": 0.0762794017791748, "sampling/importance_sampling_ratio/mean": 1.0123547315597534, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.838761143386364, "clip_ratio/low_mean": 0.08220163360238075, "clip_ratio/low_min": 0.08220163360238075, "clip_ratio/high_mean": 0.04869186133146286, "clip_ratio/high_max": 0.04869186133146286, "clip_ratio/region_mean": 0.1308934949338436, "reward_total_mean": 0.6049907207489014, "reward_meter_mean": 0.7856926918029785, "reward_meter_std": 0.21308495104312897, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9201638102531433, "reward_repeat_soft_std": 0.08705054223537445, "reward_judge_quality_mean": 0.5487500429153442, "reward_judge_quality_std": 0.22937415540218353, "reward_total_composite_mean": 0.6049907207489014, "reward_total_composite_std": 0.10036706924438477} {"timestamp_utc": "2026-04-13T10:07:40Z", "mode": "train", "global_step": 1117, "epoch": 0.11220492214967354, "loss": 0.0666, "grad_norm": 15.617884635925293, "learning_rate": 6.618181818181819e-06, "num_tokens": 1975941.0, "completions/mean_length": 36.75, "completions/min_length": 33.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.75, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.9262579083442688, "rewards/meter/std": 0.0824204757809639, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9577685594558716, "rewards/repeat_soft/std": 0.050275254994630814, "rewards/judge_quality/mean": 0.5637500286102295, "rewards/judge_quality/std": 0.22012580931186676, "rewards/total_composite/mean": 0.6820363998413086, "rewards/total_composite/std": 0.12780505418777466, "reward": 0.6820363998413086, "reward_std": 0.12780506908893585, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15699218213558197, "sampling/sampling_logp_difference/max": 5.221889495849609, "sampling/importance_sampling_ratio/min": 0.005397121887654066, "sampling/importance_sampling_ratio/mean": 1.0000742673873901, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7281496822834015, "clip_ratio/low_mean": 0.0680940211750567, "clip_ratio/low_min": 0.0680940211750567, "clip_ratio/high_mean": 0.03378378227353096, "clip_ratio/high_max": 0.03378378227353096, "clip_ratio/region_mean": 0.10187780344858766, "reward_total_mean": 0.6820363998413086, "reward_meter_mean": 0.9262579083442688, "reward_meter_std": 0.0824204757809639, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9577685594558716, "reward_repeat_soft_std": 0.050275254994630814, "reward_judge_quality_mean": 0.5637500286102295, "reward_judge_quality_std": 0.22012580931186676, "reward_total_composite_mean": 0.6820363998413086, "reward_total_composite_std": 0.12780505418777466} {"timestamp_utc": "2026-04-13T10:07:47Z", "mode": "train", "global_step": 1118, "epoch": 0.11230537418382722, "loss": 0.0051, "grad_norm": 9.490935325622559, "learning_rate": 6.615151515151516e-06, "num_tokens": 1978404.0, "completions/mean_length": 94.875, "completions/min_length": 78.0, "completions/max_length": 104.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.875, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.9267277717590332, "rewards/meter/std": 0.16889002919197083, "rewards/count_adherence/mean": 0.7749999761581421, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9018123149871826, "rewards/repeat_soft/std": 0.04975356161594391, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.5456591844558716, "rewards/total_composite/std": 0.04382103309035301, "reward": 0.5456591844558716, "reward_std": 0.04382103309035301, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12349020689725876, "sampling/sampling_logp_difference/max": 1.7906038761138916, "sampling/importance_sampling_ratio/min": 0.17763397097587585, "sampling/importance_sampling_ratio/mean": 1.005535364151001, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8234644457697868, "clip_ratio/low_mean": 0.02817411068826914, "clip_ratio/low_min": 0.02817411068826914, "clip_ratio/high_mean": 0.09194431267678738, "clip_ratio/high_max": 0.09194431267678738, "clip_ratio/region_mean": 0.12011842336505651, "reward_total_mean": 0.5456591844558716, "reward_meter_mean": 0.9267277717590332, "reward_meter_std": 0.16889002919197083, "reward_count_adherence_mean": 0.7749999761581421, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9018123149871826, "reward_repeat_soft_std": 0.04975356161594391, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.5456591844558716, "reward_total_composite_std": 0.04382103309035301} {"timestamp_utc": "2026-04-13T10:07:54Z", "mode": "train", "global_step": 1119, "epoch": 0.11240582621798091, "loss": 0.0328, "grad_norm": 12.819561004638672, "learning_rate": 6.612121212121213e-06, "num_tokens": 1980241.0, "completions/mean_length": 71.625, "completions/min_length": 60.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.625, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.9107191562652588, "rewards/meter/std": 0.2147468477487564, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8732104301452637, "rewards/repeat_soft/std": 0.06761850416660309, "rewards/judge_quality/mean": 0.4737499952316284, "rewards/judge_quality/std": 0.16291432082653046, "rewards/total_composite/mean": 0.5724388360977173, "rewards/total_composite/std": 0.13450887799263, "reward": 0.5724388360977173, "reward_std": 0.1345088630914688, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15446336567401886, "sampling/sampling_logp_difference/max": 1.814864158630371, "sampling/importance_sampling_ratio/min": 0.16286003589630127, "sampling/importance_sampling_ratio/mean": 1.038291096687317, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.326532170176506, "clip_ratio/low_mean": 0.07853138726204634, "clip_ratio/low_min": 0.07853138726204634, "clip_ratio/high_mean": 0.028238224796950817, "clip_ratio/high_max": 0.028238224796950817, "clip_ratio/region_mean": 0.10676961205899715, "reward_total_mean": 0.5724388360977173, "reward_meter_mean": 0.9107191562652588, "reward_meter_std": 0.2147468477487564, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8732104301452637, "reward_repeat_soft_std": 0.06761850416660309, "reward_judge_quality_mean": 0.4737499952316284, "reward_judge_quality_std": 0.16291432082653046, "reward_total_composite_mean": 0.5724388360977173, "reward_total_composite_std": 0.13450887799263} {"timestamp_utc": "2026-04-13T10:08:02Z", "mode": "train", "global_step": 1120, "epoch": 0.1125062782521346, "loss": 0.0482, "grad_norm": 9.773730278015137, "learning_rate": 6.609090909090909e-06, "num_tokens": 1982294.0, "completions/mean_length": 76.625, "completions/min_length": 63.0, "completions/max_length": 97.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.625, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.6077883243560791, "rewards/meter/std": 0.3168320655822754, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.837867021560669, "rewards/repeat_soft/std": 0.08495821058750153, "rewards/judge_quality/mean": 0.4362499713897705, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.46740472316741943, "rewards/total_composite/std": 0.13453896343708038, "reward": 0.46740472316741943, "reward_std": 0.13453896343708038, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14241869747638702, "sampling/sampling_logp_difference/max": 2.31327223777771, "sampling/importance_sampling_ratio/min": 0.09893697500228882, "sampling/importance_sampling_ratio/mean": 0.9904230833053589, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1188526898622513, "clip_ratio/low_mean": 0.06325502693653107, "clip_ratio/low_min": 0.06325502693653107, "clip_ratio/high_mean": 0.051562500186264515, "clip_ratio/high_max": 0.051562500186264515, "clip_ratio/region_mean": 0.11481752712279558, "reward_total_mean": 0.46740472316741943, "reward_meter_mean": 0.6077883243560791, "reward_meter_std": 0.3168320655822754, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.837867021560669, "reward_repeat_soft_std": 0.08495821058750153, "reward_judge_quality_mean": 0.4362499713897705, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.46740472316741943, "reward_total_composite_std": 0.13453896343708038} {"timestamp_utc": "2026-04-13T10:08:08Z", "mode": "train", "global_step": 1121, "epoch": 0.1126067302862883, "loss": -0.0491, "grad_norm": 10.90406608581543, "learning_rate": 6.606060606060607e-06, "num_tokens": 1984213.0, "completions/mean_length": 60.875, "completions/min_length": 47.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.875, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9739933013916016, "rewards/meter/std": 0.03326532989740372, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9222980737686157, "rewards/repeat_soft/std": 0.04028952121734619, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.5566431879997253, "rewards/total_composite/std": 0.01162195298820734, "reward": 0.5566431879997253, "reward_std": 0.011621958576142788, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1495954841375351, "sampling/sampling_logp_difference/max": 1.8567464351654053, "sampling/importance_sampling_ratio/min": 0.15617994964122772, "sampling/importance_sampling_ratio/mean": 1.0220969915390015, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1848102807998657, "clip_ratio/low_mean": 0.03970585763454437, "clip_ratio/low_min": 0.03970585763454437, "clip_ratio/high_mean": 0.08167519699782133, "clip_ratio/high_max": 0.08167519699782133, "clip_ratio/region_mean": 0.1213810546323657, "reward_total_mean": 0.5566431879997253, "reward_meter_mean": 0.9739933013916016, "reward_meter_std": 0.03326532989740372, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9222980737686157, "reward_repeat_soft_std": 0.04028952121734619, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.5566431879997253, "reward_total_composite_std": 0.01162195298820734} {"timestamp_utc": "2026-04-13T10:08:15Z", "mode": "train", "global_step": 1122, "epoch": 0.112707182320442, "loss": -0.0273, "grad_norm": 8.987536430358887, "learning_rate": 6.603030303030303e-06, "num_tokens": 1986309.0, "completions/mean_length": 77.0, "completions/min_length": 68.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.0, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.8478034734725952, "rewards/meter/std": 0.17993468046188354, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.94336998462677, "rewards/repeat_soft/std": 0.024993643164634705, "rewards/judge_quality/mean": 0.38624998927116394, "rewards/judge_quality/std": 0.09545940905809402, "rewards/total_composite/mean": 0.5010881423950195, "rewards/total_composite/std": 0.06411530077457428, "reward": 0.5010881423950195, "reward_std": 0.06411530822515488, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1549995243549347, "sampling/sampling_logp_difference/max": 2.250798225402832, "sampling/importance_sampling_ratio/min": 0.10531511902809143, "sampling/importance_sampling_ratio/mean": 1.0179111957550049, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4146091043949127, "clip_ratio/low_mean": 0.05337740574032068, "clip_ratio/low_min": 0.05337740574032068, "clip_ratio/high_mean": 0.06918917689472437, "clip_ratio/high_max": 0.06918917689472437, "clip_ratio/region_mean": 0.12256658263504505, "reward_total_mean": 0.5010881423950195, "reward_meter_mean": 0.8478034734725952, "reward_meter_std": 0.17993468046188354, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.94336998462677, "reward_repeat_soft_std": 0.024993643164634705, "reward_judge_quality_mean": 0.38624998927116394, "reward_judge_quality_std": 0.09545940905809402, "reward_total_composite_mean": 0.5010881423950195, "reward_total_composite_std": 0.06411530077457428} {"timestamp_utc": "2026-04-13T10:08:21Z", "mode": "train", "global_step": 1123, "epoch": 0.11280763435459568, "loss": 0.0172, "grad_norm": 11.716697692871094, "learning_rate": 6.600000000000001e-06, "num_tokens": 1988086.0, "completions/mean_length": 55.125, "completions/min_length": 47.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.125, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9374167323112488, "rewards/meter/std": 0.10473205149173737, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9076154232025146, "rewards/repeat_soft/std": 0.08004806190729141, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.5522748231887817, "rewards/total_composite/std": 0.0624866783618927, "reward": 0.5522748231887817, "reward_std": 0.0624866746366024, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1449204832315445, "sampling/sampling_logp_difference/max": 1.5615653991699219, "sampling/importance_sampling_ratio/min": 0.2098073810338974, "sampling/importance_sampling_ratio/mean": 1.012521505355835, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2464976161718369, "clip_ratio/low_mean": 0.04081095149740577, "clip_ratio/low_min": 0.04081095149740577, "clip_ratio/high_mean": 0.07589972019195557, "clip_ratio/high_max": 0.07589972019195557, "clip_ratio/region_mean": 0.11671067168936133, "reward_total_mean": 0.5522748231887817, "reward_meter_mean": 0.9374167323112488, "reward_meter_std": 0.10473205149173737, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9076154232025146, "reward_repeat_soft_std": 0.08004806190729141, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.5522748231887817, "reward_total_composite_std": 0.0624866783618927} {"timestamp_utc": "2026-04-13T10:08:28Z", "mode": "train", "global_step": 1124, "epoch": 0.11290808638874937, "loss": 0.0262, "grad_norm": 9.961459159851074, "learning_rate": 6.596969696969698e-06, "num_tokens": 1990035.0, "completions/mean_length": 62.625, "completions/min_length": 54.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.625, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.7475841045379639, "rewards/meter/std": 0.19611549377441406, "rewards/count_adherence/mean": 0.7708333134651184, "rewards/count_adherence/std": 0.0862581729888916, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8471733927726746, "rewards/repeat_soft/std": 0.07111194729804993, "rewards/judge_quality/mean": 0.4362499713897705, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.4932330250740051, "rewards/total_composite/std": 0.08194217085838318, "reward": 0.4932330250740051, "reward_std": 0.08194217085838318, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10107006132602692, "sampling/sampling_logp_difference/max": 3.2778851985931396, "sampling/importance_sampling_ratio/min": 0.236572265625, "sampling/importance_sampling_ratio/mean": 1.0227466821670532, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5160208456218243, "clip_ratio/low_mean": 0.03161199390888214, "clip_ratio/low_min": 0.03161199390888214, "clip_ratio/high_mean": 0.06855028681457043, "clip_ratio/high_max": 0.06855028681457043, "clip_ratio/region_mean": 0.10016228072345257, "reward_total_mean": 0.4932330250740051, "reward_meter_mean": 0.7475841045379639, "reward_meter_std": 0.19611549377441406, "reward_count_adherence_mean": 0.7708333134651184, "reward_count_adherence_std": 0.0862581729888916, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8471733927726746, "reward_repeat_soft_std": 0.07111194729804993, "reward_judge_quality_mean": 0.4362499713897705, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.4932330250740051, "reward_total_composite_std": 0.08194217085838318} {"timestamp_utc": "2026-04-13T10:08:34Z", "mode": "train", "global_step": 1125, "epoch": 0.11300853842290307, "loss": -0.0262, "grad_norm": 7.455492973327637, "learning_rate": 6.593939393939395e-06, "num_tokens": 1991683.0, "completions/mean_length": 41.0, "completions/min_length": 37.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.0, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.8595741391181946, "rewards/meter/std": 0.3456464409828186, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8770755529403687, "rewards/repeat_soft/std": 0.11364620178937912, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.5710352659225464, "rewards/total_composite/std": 0.11277426779270172, "reward": 0.5710352659225464, "reward_std": 0.11277425289154053, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12854094803333282, "sampling/sampling_logp_difference/max": 1.1401591300964355, "sampling/importance_sampling_ratio/min": 0.31976813077926636, "sampling/importance_sampling_ratio/mean": 1.043609380722046, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0445300117135048, "clip_ratio/low_mean": 0.013513513840734959, "clip_ratio/low_min": 0.013513513840734959, "clip_ratio/high_mean": 0.12344709038734436, "clip_ratio/high_max": 0.12344709038734436, "clip_ratio/region_mean": 0.13696060422807932, "reward_total_mean": 0.5710352659225464, "reward_meter_mean": 0.8595741391181946, "reward_meter_std": 0.3456464409828186, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8770755529403687, "reward_repeat_soft_std": 0.11364620178937912, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.5710352659225464, "reward_total_composite_std": 0.11277426779270172} {"timestamp_utc": "2026-04-13T10:08:40Z", "mode": "train", "global_step": 1126, "epoch": 0.11310899045705676, "loss": -0.0292, "grad_norm": 14.455120086669922, "learning_rate": 6.590909090909091e-06, "num_tokens": 1993459.0, "completions/mean_length": 42.0, "completions/min_length": 34.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.0, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.9213700294494629, "rewards/meter/std": 0.11122234910726547, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9119148850440979, "rewards/repeat_soft/std": 0.0637424886226654, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.6664919257164001, "rewards/total_composite/std": 0.14875824749469757, "reward": 0.6664919257164001, "reward_std": 0.14875824749469757, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14410021901130676, "sampling/sampling_logp_difference/max": 1.0909414291381836, "sampling/importance_sampling_ratio/min": 0.33590012788772583, "sampling/importance_sampling_ratio/mean": 1.0286000967025757, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.042415238916874, "clip_ratio/low_mean": 0.09365024883300066, "clip_ratio/low_min": 0.09365024883300066, "clip_ratio/high_mean": 0.022001934237778187, "clip_ratio/high_max": 0.022001934237778187, "clip_ratio/region_mean": 0.11565218307077885, "reward_total_mean": 0.6664919257164001, "reward_meter_mean": 0.9213700294494629, "reward_meter_std": 0.11122234910726547, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9119148850440979, "reward_repeat_soft_std": 0.0637424886226654, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.6664919257164001, "reward_total_composite_std": 0.14875824749469757} {"timestamp_utc": "2026-04-13T10:08:51Z", "mode": "train", "global_step": 1127, "epoch": 0.11320944249121044, "loss": -0.0753, "grad_norm": 2.7037055492401123, "learning_rate": 6.5878787878787885e-06, "num_tokens": 1994871.0, "completions/mean_length": 84.5, "completions/min_length": 20.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 23.428571701049805, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.7526909112930298, "rewards/meter/std": 0.36496224999427795, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9083139300346375, "rewards/repeat_soft/std": 0.07009585946798325, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.1528538018465042, "rewards/total_composite/mean": 0.4741763472557068, "rewards/total_composite/std": 0.2111952304840088, "reward": 0.4741763472557068, "reward_std": 0.2111952304840088, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13462936878204346, "sampling/sampling_logp_difference/max": 0.811007022857666, "sampling/importance_sampling_ratio/min": 0.45685768127441406, "sampling/importance_sampling_ratio/mean": 1.0266838073730469, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.125832237303257, "clip_ratio/low_mean": 0.024038461968302727, "clip_ratio/low_min": 0.024038461968302727, "clip_ratio/high_mean": 0.11665728315711021, "clip_ratio/high_max": 0.11665728315711021, "clip_ratio/region_mean": 0.14069574512541294, "reward_total_mean": 0.4741763472557068, "reward_meter_mean": 0.7526909112930298, "reward_meter_std": 0.36496224999427795, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9083139300346375, "reward_repeat_soft_std": 0.07009585946798325, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.1528538018465042, "reward_total_composite_mean": 0.4741763472557068, "reward_total_composite_std": 0.2111952304840088} {"timestamp_utc": "2026-04-13T10:09:02Z", "mode": "train", "global_step": 1128, "epoch": 0.11330989452536414, "loss": -0.0627, "grad_norm": 3.1919827461242676, "learning_rate": 6.584848484848485e-06, "num_tokens": 1996325.0, "completions/mean_length": 91.75, "completions/min_length": 26.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 31.71428680419922, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.6264134049415588, "rewards/meter/std": 0.4235175848007202, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.998651385307312, "rewards/repeat_soft/std": 0.003814458381384611, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.23445606231689453, "rewards/total_composite/mean": 0.5159874558448792, "rewards/total_composite/std": 0.2651650309562683, "reward": 0.5159874558448792, "reward_std": 0.2651650309562683, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14543375372886658, "sampling/sampling_logp_difference/max": 1.6496882438659668, "sampling/importance_sampling_ratio/min": 0.19210979342460632, "sampling/importance_sampling_ratio/mean": 1.036872386932373, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6218914948403835, "clip_ratio/low_mean": 0.036608658730983734, "clip_ratio/low_min": 0.036608658730983734, "clip_ratio/high_mean": 0.09086930099874735, "clip_ratio/high_max": 0.09086930099874735, "clip_ratio/region_mean": 0.12747795972973108, "reward_total_mean": 0.5159874558448792, "reward_meter_mean": 0.6264134049415588, "reward_meter_std": 0.4235175848007202, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.998651385307312, "reward_repeat_soft_std": 0.003814458381384611, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.23445606231689453, "reward_total_composite_mean": 0.5159874558448792, "reward_total_composite_std": 0.2651650309562683} {"timestamp_utc": "2026-04-13T10:09:13Z", "mode": "train", "global_step": 1129, "epoch": 0.11341034655951783, "loss": -0.1115, "grad_norm": 2.871786117553711, "learning_rate": 6.581818181818182e-06, "num_tokens": 1997814.0, "completions/mean_length": 98.125, "completions/min_length": 33.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 39.0, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.9012106657028198, "rewards/meter/std": 0.17986968159675598, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9321990609169006, "rewards/repeat_soft/std": 0.03417094051837921, "rewards/judge_quality/mean": 0.4012500047683716, "rewards/judge_quality/std": 0.19773268699645996, "rewards/total_composite/mean": 0.5356782078742981, "rewards/total_composite/std": 0.24061591923236847, "reward": 0.5356782078742981, "reward_std": 0.24061590433120728, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1484362781047821, "sampling/sampling_logp_difference/max": 1.0686688423156738, "sampling/importance_sampling_ratio/min": 0.3434654176235199, "sampling/importance_sampling_ratio/mean": 1.011346459388733, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0026362836360931, "clip_ratio/low_mean": 0.013888888992369175, "clip_ratio/low_min": 0.013888888992369175, "clip_ratio/high_mean": 0.08541359659284353, "clip_ratio/high_max": 0.08541359659284353, "clip_ratio/region_mean": 0.09930248558521271, "reward_total_mean": 0.5356782078742981, "reward_meter_mean": 0.9012106657028198, "reward_meter_std": 0.17986968159675598, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9321990609169006, "reward_repeat_soft_std": 0.03417094051837921, "reward_judge_quality_mean": 0.4012500047683716, "reward_judge_quality_std": 0.19773268699645996, "reward_total_composite_mean": 0.5356782078742981, "reward_total_composite_std": 0.24061591923236847} {"timestamp_utc": "2026-04-13T10:09:19Z", "mode": "train", "global_step": 1130, "epoch": 0.11351079859367152, "loss": 0.0142, "grad_norm": 9.468161582946777, "learning_rate": 6.578787878787879e-06, "num_tokens": 1999386.0, "completions/mean_length": 44.5, "completions/min_length": 43.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.5, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.5640295147895813, "rewards/meter/std": 0.34981825947761536, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8968439102172852, "rewards/repeat_soft/std": 0.054242637008428574, "rewards/judge_quality/mean": 0.7325000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.5763897895812988, "rewards/total_composite/std": 0.17131243646144867, "reward": 0.5763897895812988, "reward_std": 0.17131242156028748, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10996052622795105, "sampling/sampling_logp_difference/max": 2.1871838569641113, "sampling/importance_sampling_ratio/min": 0.11223236471414566, "sampling/importance_sampling_ratio/mean": 0.9965682625770569, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5740429349243641, "clip_ratio/low_mean": 0.05572951911017299, "clip_ratio/low_min": 0.05572951911017299, "clip_ratio/high_mean": 0.025766385719180107, "clip_ratio/high_max": 0.025766385719180107, "clip_ratio/region_mean": 0.0814959048293531, "reward_total_mean": 0.5763897895812988, "reward_meter_mean": 0.5640295147895813, "reward_meter_std": 0.34981825947761536, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8968439102172852, "reward_repeat_soft_std": 0.054242637008428574, "reward_judge_quality_mean": 0.7325000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.5763897895812988, "reward_total_composite_std": 0.17131243646144867} {"timestamp_utc": "2026-04-13T10:09:27Z", "mode": "train", "global_step": 1131, "epoch": 0.11361125062782522, "loss": -0.1178, "grad_norm": 14.717326164245605, "learning_rate": 6.575757575757577e-06, "num_tokens": 2001222.0, "completions/mean_length": 63.5, "completions/min_length": 41.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.5, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.8636382818222046, "rewards/meter/std": 0.2630917727947235, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9741989374160767, "rewards/repeat_soft/std": 0.04893767088651657, "rewards/judge_quality/mean": 0.6575000286102295, "rewards/judge_quality/std": 0.20658794045448303, "rewards/total_composite/mean": 0.7093788385391235, "rewards/total_composite/std": 0.18725824356079102, "reward": 0.7093788385391235, "reward_std": 0.18725824356079102, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1483689695596695, "sampling/sampling_logp_difference/max": 2.488309860229492, "sampling/importance_sampling_ratio/min": 0.0830502137541771, "sampling/importance_sampling_ratio/mean": 1.0160596370697021, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6833672076463699, "clip_ratio/low_mean": 0.05194827914237976, "clip_ratio/low_min": 0.05194827914237976, "clip_ratio/high_mean": 0.07146300747990608, "clip_ratio/high_max": 0.07146300747990608, "clip_ratio/region_mean": 0.12341128662228584, "reward_total_mean": 0.7093788385391235, "reward_meter_mean": 0.8636382818222046, "reward_meter_std": 0.2630917727947235, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9741989374160767, "reward_repeat_soft_std": 0.04893767088651657, "reward_judge_quality_mean": 0.6575000286102295, "reward_judge_quality_std": 0.20658794045448303, "reward_total_composite_mean": 0.7093788385391235, "reward_total_composite_std": 0.18725824356079102} {"timestamp_utc": "2026-04-13T10:09:33Z", "mode": "train", "global_step": 1132, "epoch": 0.1137117026619789, "loss": 0.0658, "grad_norm": 12.287823677062988, "learning_rate": 6.572727272727273e-06, "num_tokens": 2002814.0, "completions/mean_length": 40.0, "completions/min_length": 34.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.0, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9884093999862671, "rewards/meter/std": 0.006536794826388359, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9248701333999634, "rewards/repeat_soft/std": 0.06530743837356567, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.61822110414505, "rewards/total_composite/std": 0.017233116552233696, "reward": 0.61822110414505, "reward_std": 0.01723311096429825, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14291036128997803, "sampling/sampling_logp_difference/max": 2.1756439208984375, "sampling/importance_sampling_ratio/min": 0.11353502422571182, "sampling/importance_sampling_ratio/mean": 1.0182663202285767, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1009890586137772, "clip_ratio/low_mean": 0.06933821551501751, "clip_ratio/low_min": 0.06933821551501751, "clip_ratio/high_mean": 0.07251258380711079, "clip_ratio/high_max": 0.07251258380711079, "clip_ratio/region_mean": 0.1418507993221283, "reward_total_mean": 0.61822110414505, "reward_meter_mean": 0.9884093999862671, "reward_meter_std": 0.006536794826388359, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9248701333999634, "reward_repeat_soft_std": 0.06530743837356567, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.61822110414505, "reward_total_composite_std": 0.017233116552233696} {"timestamp_utc": "2026-04-13T10:09:45Z", "mode": "train", "global_step": 1133, "epoch": 0.1138121546961326, "loss": -0.1135, "grad_norm": 2.8444020748138428, "learning_rate": 6.56969696969697e-06, "num_tokens": 2004482.0, "completions/mean_length": 102.5, "completions/min_length": 39.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 44.000003814697266, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.6006203293800354, "rewards/meter/std": 0.4255194664001465, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9695420861244202, "rewards/repeat_soft/std": 0.0406782403588295, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.13617216050624847, "rewards/total_composite/mean": 0.4698367714881897, "rewards/total_composite/std": 0.21603558957576752, "reward": 0.4698367714881897, "reward_std": 0.21603557467460632, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1557750254869461, "sampling/sampling_logp_difference/max": 1.7177009582519531, "sampling/importance_sampling_ratio/min": 0.17947831749916077, "sampling/importance_sampling_ratio/mean": 1.0362001657485962, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0773121491074562, "clip_ratio/low_mean": 0.037367021664977074, "clip_ratio/low_min": 0.037367021664977074, "clip_ratio/high_mean": 0.07085578422993422, "clip_ratio/high_max": 0.07085578422993422, "clip_ratio/region_mean": 0.10822280589491129, "reward_total_mean": 0.4698367714881897, "reward_meter_mean": 0.6006203293800354, "reward_meter_std": 0.4255194664001465, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9695420861244202, "reward_repeat_soft_std": 0.0406782403588295, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.13617216050624847, "reward_total_composite_mean": 0.4698367714881897, "reward_total_composite_std": 0.21603558957576752} {"timestamp_utc": "2026-04-13T10:09:51Z", "mode": "train", "global_step": 1134, "epoch": 0.11391260673028629, "loss": 0.0101, "grad_norm": 10.173377990722656, "learning_rate": 6.566666666666667e-06, "num_tokens": 2006521.0, "completions/mean_length": 66.875, "completions/min_length": 54.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.875, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.9178117513656616, "rewards/meter/std": 0.11717532575130463, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9686312675476074, "rewards/repeat_soft/std": 0.019978869706392288, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.5446088910102844, "rewards/total_composite/std": 0.05727742984890938, "reward": 0.5446088910102844, "reward_std": 0.05727742612361908, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18602125346660614, "sampling/sampling_logp_difference/max": 1.8618534803390503, "sampling/importance_sampling_ratio/min": 0.155384361743927, "sampling/importance_sampling_ratio/mean": 1.010296106338501, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7691242843866348, "clip_ratio/low_mean": 0.0799316605553031, "clip_ratio/low_min": 0.0799316605553031, "clip_ratio/high_mean": 0.06743836030364037, "clip_ratio/high_max": 0.06743836030364037, "clip_ratio/region_mean": 0.14737002085894346, "reward_total_mean": 0.5446088910102844, "reward_meter_mean": 0.9178117513656616, "reward_meter_std": 0.11717532575130463, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9686312675476074, "reward_repeat_soft_std": 0.019978869706392288, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.5446088910102844, "reward_total_composite_std": 0.05727742984890938} {"timestamp_utc": "2026-04-13T10:09:58Z", "mode": "train", "global_step": 1135, "epoch": 0.11401305876443998, "loss": 0.0051, "grad_norm": 10.730149269104004, "learning_rate": 6.563636363636364e-06, "num_tokens": 2008640.0, "completions/mean_length": 80.875, "completions/min_length": 76.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.875, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.5640273690223694, "rewards/meter/std": 0.45509153604507446, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9958585500717163, "rewards/repeat_soft/std": 0.0028345701284706593, "rewards/judge_quality/mean": 0.5024999976158142, "rewards/judge_quality/std": 0.1440981924533844, "rewards/total_composite/mean": 0.50409996509552, "rewards/total_composite/std": 0.19318099319934845, "reward": 0.50409996509552, "reward_std": 0.19318099319934845, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15317434072494507, "sampling/sampling_logp_difference/max": 1.7908809185028076, "sampling/importance_sampling_ratio/min": 0.16681316494941711, "sampling/importance_sampling_ratio/mean": 1.0111745595932007, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2982793375849724, "clip_ratio/low_mean": 0.07364986091852188, "clip_ratio/low_min": 0.07364986091852188, "clip_ratio/high_mean": 0.07550835236907005, "clip_ratio/high_max": 0.07550835236907005, "clip_ratio/region_mean": 0.14915821328759193, "reward_total_mean": 0.50409996509552, "reward_meter_mean": 0.5640273690223694, "reward_meter_std": 0.45509153604507446, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9958585500717163, "reward_repeat_soft_std": 0.0028345701284706593, "reward_judge_quality_mean": 0.5024999976158142, "reward_judge_quality_std": 0.1440981924533844, "reward_total_composite_mean": 0.50409996509552, "reward_total_composite_std": 0.19318099319934845} {"timestamp_utc": "2026-04-13T10:10:04Z", "mode": "train", "global_step": 1136, "epoch": 0.11411351079859366, "loss": 0.0649, "grad_norm": 14.582271575927734, "learning_rate": 6.56060606060606e-06, "num_tokens": 2010272.0, "completions/mean_length": 46.0, "completions/min_length": 37.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.0, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9082609415054321, "rewards/meter/std": 0.18846864998340607, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9454102516174316, "rewards/repeat_soft/std": 0.04497417062520981, "rewards/judge_quality/mean": 0.690000057220459, "rewards/judge_quality/std": 0.22315914928913116, "rewards/total_composite/mean": 0.7473007440567017, "rewards/total_composite/std": 0.1645580530166626, "reward": 0.7473007440567017, "reward_std": 0.1645580530166626, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1381937712430954, "sampling/sampling_logp_difference/max": 1.9469313621520996, "sampling/importance_sampling_ratio/min": 0.1427113264799118, "sampling/importance_sampling_ratio/mean": 1.019103765487671, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0522950664162636, "clip_ratio/low_mean": 0.051837935112416744, "clip_ratio/low_min": 0.051837935112416744, "clip_ratio/high_mean": 0.08532747812569141, "clip_ratio/high_max": 0.08532747812569141, "clip_ratio/region_mean": 0.13716541323810816, "reward_total_mean": 0.7473007440567017, "reward_meter_mean": 0.9082609415054321, "reward_meter_std": 0.18846864998340607, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9454102516174316, "reward_repeat_soft_std": 0.04497417062520981, "reward_judge_quality_mean": 0.690000057220459, "reward_judge_quality_std": 0.22315914928913116, "reward_total_composite_mean": 0.7473007440567017, "reward_total_composite_std": 0.1645580530166626} {"timestamp_utc": "2026-04-13T10:10:15Z", "mode": "train", "global_step": 1137, "epoch": 0.11421396283274736, "loss": -0.1119, "grad_norm": 3.1062843799591064, "learning_rate": 6.5575757575757585e-06, "num_tokens": 2011803.0, "completions/mean_length": 100.375, "completions/min_length": 38.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 41.57143020629883, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.8784385919570923, "rewards/meter/std": 0.2813236713409424, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9840956926345825, "rewards/repeat_soft/std": 0.015371386893093586, "rewards/judge_quality/mean": 0.45249998569488525, "rewards/judge_quality/std": 0.2665520906448364, "rewards/total_composite/mean": 0.5879136323928833, "rewards/total_composite/std": 0.2742556035518646, "reward": 0.5879136323928833, "reward_std": 0.2742556035518646, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13355565071105957, "sampling/sampling_logp_difference/max": 1.1007308959960938, "sampling/importance_sampling_ratio/min": 0.33262789249420166, "sampling/importance_sampling_ratio/mean": 1.0302464962005615, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0896887928247452, "clip_ratio/low_mean": 0.00937500037252903, "clip_ratio/low_min": 0.00937500037252903, "clip_ratio/high_mean": 0.10518785053864121, "clip_ratio/high_max": 0.10518785053864121, "clip_ratio/region_mean": 0.11456285091117024, "reward_total_mean": 0.5879136323928833, "reward_meter_mean": 0.8784385919570923, "reward_meter_std": 0.2813236713409424, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9840956926345825, "reward_repeat_soft_std": 0.015371386893093586, "reward_judge_quality_mean": 0.45249998569488525, "reward_judge_quality_std": 0.2665520906448364, "reward_total_composite_mean": 0.5879136323928833, "reward_total_composite_std": 0.2742556035518646} {"timestamp_utc": "2026-04-13T10:10:21Z", "mode": "train", "global_step": 1138, "epoch": 0.11431441486690105, "loss": 0.0294, "grad_norm": 14.06969928741455, "learning_rate": 6.554545454545455e-06, "num_tokens": 2013311.0, "completions/mean_length": 31.5, "completions/min_length": 28.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.5, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.4891081750392914, "rewards/meter/std": 0.3816754221916199, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9245496988296509, "rewards/repeat_soft/std": 0.07919017225503922, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.5305708050727844, "rewards/total_composite/std": 0.17777620255947113, "reward": 0.5305708050727844, "reward_std": 0.17777620255947113, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14619868993759155, "sampling/sampling_logp_difference/max": 1.7692971229553223, "sampling/importance_sampling_ratio/min": 0.17045274376869202, "sampling/importance_sampling_ratio/mean": 0.9935710430145264, "sampling/importance_sampling_ratio/max": 1.962249755859375, "entropy": 0.8535003513097763, "clip_ratio/low_mean": 0.06597222480922937, "clip_ratio/low_min": 0.06597222480922937, "clip_ratio/high_mean": 0.06373106129467487, "clip_ratio/high_max": 0.06373106129467487, "clip_ratio/region_mean": 0.12970328610390425, "reward_total_mean": 0.5305708050727844, "reward_meter_mean": 0.4891081750392914, "reward_meter_std": 0.3816754221916199, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9245496988296509, "reward_repeat_soft_std": 0.07919017225503922, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.5305708050727844, "reward_total_composite_std": 0.17777620255947113} {"timestamp_utc": "2026-04-13T10:10:32Z", "mode": "train", "global_step": 1139, "epoch": 0.11441486690105475, "loss": -0.1127, "grad_norm": 1.582905650138855, "learning_rate": 6.551515151515152e-06, "num_tokens": 2014823.0, "completions/mean_length": 95.0, "completions/min_length": 32.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 35.42857360839844, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.909144401550293, "rewards/meter/std": 0.22075851261615753, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9273666143417358, "rewards/repeat_soft/std": 0.059534281492233276, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.13617216050624847, "rewards/total_composite/mean": 0.53968346118927, "rewards/total_composite/std": 0.21839000284671783, "reward": 0.53968346118927, "reward_std": 0.21838998794555664, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11558738350868225, "sampling/sampling_logp_difference/max": 1.6367340087890625, "sampling/importance_sampling_ratio/min": 0.19461461901664734, "sampling/importance_sampling_ratio/mean": 1.0194705724716187, "sampling/importance_sampling_ratio/max": 1.8529249429702759, "entropy": 0.665039174258709, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.08882309589534998, "clip_ratio/high_max": 0.08882309589534998, "clip_ratio/region_mean": 0.08882309589534998, "reward_total_mean": 0.53968346118927, "reward_meter_mean": 0.909144401550293, "reward_meter_std": 0.22075851261615753, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9273666143417358, "reward_repeat_soft_std": 0.059534281492233276, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.13617216050624847, "reward_total_composite_mean": 0.53968346118927, "reward_total_composite_std": 0.21839000284671783} {"timestamp_utc": "2026-04-13T10:10:44Z", "mode": "train", "global_step": 1140, "epoch": 0.11451531893520844, "loss": -0.1527, "grad_norm": 3.4708738327026367, "learning_rate": 6.5484848484848494e-06, "num_tokens": 2016805.0, "completions/mean_length": 115.75, "completions/min_length": 49.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 59.142860412597656, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.35854005813598633, "rewards/meter/std": 0.371934711933136, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9448579549789429, "rewards/repeat_soft/std": 0.04365341365337372, "rewards/judge_quality/mean": 0.41874998807907104, "rewards/judge_quality/std": 0.18074746429920197, "rewards/total_composite/mean": 0.35877785086631775, "rewards/total_composite/std": 0.17009501159191132, "reward": 0.35877785086631775, "reward_std": 0.17009499669075012, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.133514866232872, "sampling/sampling_logp_difference/max": 1.222212791442871, "sampling/importance_sampling_ratio/min": 0.29457759857177734, "sampling/importance_sampling_ratio/mean": 1.0175260305404663, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8386684954166412, "clip_ratio/low_mean": 0.05946615990251303, "clip_ratio/low_min": 0.05946615990251303, "clip_ratio/high_mean": 0.06686240807175636, "clip_ratio/high_max": 0.06686240807175636, "clip_ratio/region_mean": 0.1263285679742694, "reward_total_mean": 0.35877785086631775, "reward_meter_mean": 0.35854005813598633, "reward_meter_std": 0.371934711933136, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9448579549789429, "reward_repeat_soft_std": 0.04365341365337372, "reward_judge_quality_mean": 0.41874998807907104, "reward_judge_quality_std": 0.18074746429920197, "reward_total_composite_mean": 0.35877785086631775, "reward_total_composite_std": 0.17009501159191132} {"timestamp_utc": "2026-04-13T10:10:50Z", "mode": "train", "global_step": 1141, "epoch": 0.11461577096936212, "loss": 0.039, "grad_norm": 12.786428451538086, "learning_rate": 6.545454545454546e-06, "num_tokens": 2018367.0, "completions/mean_length": 41.25, "completions/min_length": 39.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.25, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.6099901795387268, "rewards/meter/std": 0.3610473871231079, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8980588912963867, "rewards/repeat_soft/std": 0.08926127105951309, "rewards/judge_quality/mean": 0.6150000095367432, "rewards/judge_quality/std": 0.25286927819252014, "rewards/total_composite/mean": 0.574455976486206, "rewards/total_composite/std": 0.16856855154037476, "reward": 0.574455976486206, "reward_std": 0.16856855154037476, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14846009016036987, "sampling/sampling_logp_difference/max": 1.8245267868041992, "sampling/importance_sampling_ratio/min": 0.16129395365715027, "sampling/importance_sampling_ratio/mean": 1.027154564857483, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.111771211028099, "clip_ratio/low_mean": 0.056686026975512505, "clip_ratio/low_min": 0.056686026975512505, "clip_ratio/high_mean": 0.06761363800615072, "clip_ratio/high_max": 0.06761363800615072, "clip_ratio/region_mean": 0.12429966498166323, "reward_total_mean": 0.574455976486206, "reward_meter_mean": 0.6099901795387268, "reward_meter_std": 0.3610473871231079, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8980588912963867, "reward_repeat_soft_std": 0.08926127105951309, "reward_judge_quality_mean": 0.6150000095367432, "reward_judge_quality_std": 0.25286927819252014, "reward_total_composite_mean": 0.574455976486206, "reward_total_composite_std": 0.16856855154037476} {"timestamp_utc": "2026-04-13T10:11:00Z", "mode": "train", "global_step": 1142, "epoch": 0.11471622300351582, "loss": -0.0552, "grad_norm": 12.804361343383789, "learning_rate": 6.542424242424243e-06, "num_tokens": 2020081.0, "completions/mean_length": 52.25, "completions/min_length": 43.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.25, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.8938835859298706, "rewards/meter/std": 0.09182074666023254, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.895793080329895, "rewards/repeat_soft/std": 0.048144880682229996, "rewards/judge_quality/mean": 0.5237500071525574, "rewards/judge_quality/std": 0.19078317284584045, "rewards/total_composite/mean": 0.6230459213256836, "rewards/total_composite/std": 0.13977524638175964, "reward": 0.6230459213256836, "reward_std": 0.13977523148059845, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12277662754058838, "sampling/sampling_logp_difference/max": 1.502573013305664, "sampling/importance_sampling_ratio/min": 0.22255678474903107, "sampling/importance_sampling_ratio/mean": 1.0013785362243652, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6223147809505463, "clip_ratio/low_mean": 0.07965277880430222, "clip_ratio/low_min": 0.07965277880430222, "clip_ratio/high_mean": 0.02281746082007885, "clip_ratio/high_max": 0.02281746082007885, "clip_ratio/region_mean": 0.10247023962438107, "reward_total_mean": 0.6230459213256836, "reward_meter_mean": 0.8938835859298706, "reward_meter_std": 0.09182074666023254, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.895793080329895, "reward_repeat_soft_std": 0.048144880682229996, "reward_judge_quality_mean": 0.5237500071525574, "reward_judge_quality_std": 0.19078317284584045, "reward_total_composite_mean": 0.6230459213256836, "reward_total_composite_std": 0.13977524638175964} {"timestamp_utc": "2026-04-13T10:11:06Z", "mode": "train", "global_step": 1143, "epoch": 0.11481667503766951, "loss": -0.0002, "grad_norm": 15.238978385925293, "learning_rate": 6.5393939393939395e-06, "num_tokens": 2021646.0, "completions/mean_length": 41.625, "completions/min_length": 33.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.625, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.9243209958076477, "rewards/meter/std": 0.10788122564554214, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.989984393119812, "rewards/repeat_soft/std": 0.01322022546082735, "rewards/judge_quality/mean": 0.5987499952316284, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.7137107849121094, "rewards/total_composite/std": 0.1404464691877365, "reward": 0.7137107849121094, "reward_std": 0.14044643938541412, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15316607058048248, "sampling/sampling_logp_difference/max": 1.1398444175720215, "sampling/importance_sampling_ratio/min": 0.319868803024292, "sampling/importance_sampling_ratio/mean": 1.0187184810638428, "sampling/importance_sampling_ratio/max": 1.9264450073242188, "entropy": 1.0886655524373055, "clip_ratio/low_mean": 0.06184093654155731, "clip_ratio/low_min": 0.06184093654155731, "clip_ratio/high_mean": 0.0540414759889245, "clip_ratio/high_max": 0.0540414759889245, "clip_ratio/region_mean": 0.11588241253048182, "reward_total_mean": 0.7137107849121094, "reward_meter_mean": 0.9243209958076477, "reward_meter_std": 0.10788122564554214, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.989984393119812, "reward_repeat_soft_std": 0.01322022546082735, "reward_judge_quality_mean": 0.5987499952316284, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.7137107849121094, "reward_total_composite_std": 0.1404464691877365} {"timestamp_utc": "2026-04-13T10:11:13Z", "mode": "train", "global_step": 1144, "epoch": 0.11491712707182321, "loss": 0.1071, "grad_norm": 9.581685066223145, "learning_rate": 6.536363636363638e-06, "num_tokens": 2023599.0, "completions/mean_length": 71.125, "completions/min_length": 52.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.125, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.9341791868209839, "rewards/meter/std": 0.11174824833869934, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9516667127609253, "rewards/repeat_soft/std": 0.028464632108807564, "rewards/judge_quality/mean": 0.5575000047683716, "rewards/judge_quality/std": 0.19955310225486755, "rewards/total_composite/mean": 0.6210188269615173, "rewards/total_composite/std": 0.08535400032997131, "reward": 0.6210188269615173, "reward_std": 0.08535400032997131, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14601223170757294, "sampling/sampling_logp_difference/max": 1.1267385482788086, "sampling/importance_sampling_ratio/min": 0.32408854365348816, "sampling/importance_sampling_ratio/mean": 1.020800232887268, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2325196266174316, "clip_ratio/low_mean": 0.08254536800086498, "clip_ratio/low_min": 0.08254536800086498, "clip_ratio/high_mean": 0.06810528226196766, "clip_ratio/high_max": 0.06810528226196766, "clip_ratio/region_mean": 0.15065065026283264, "reward_total_mean": 0.6210188269615173, "reward_meter_mean": 0.9341791868209839, "reward_meter_std": 0.11174824833869934, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9516667127609253, "reward_repeat_soft_std": 0.028464632108807564, "reward_judge_quality_mean": 0.5575000047683716, "reward_judge_quality_std": 0.19955310225486755, "reward_total_composite_mean": 0.6210188269615173, "reward_total_composite_std": 0.08535400032997131} {"timestamp_utc": "2026-04-13T10:11:20Z", "mode": "train", "global_step": 1145, "epoch": 0.1150175791059769, "loss": 0.0505, "grad_norm": 10.74338436126709, "learning_rate": 6.533333333333334e-06, "num_tokens": 2025811.0, "completions/mean_length": 77.5, "completions/min_length": 68.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.5, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.8966401815414429, "rewards/meter/std": 0.14012953639030457, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9699361324310303, "rewards/repeat_soft/std": 0.021746700629591942, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.13695022463798523, "rewards/total_composite/mean": 0.5656353831291199, "rewards/total_composite/std": 0.10508348792791367, "reward": 0.5656353831291199, "reward_std": 0.10508348792791367, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1363961398601532, "sampling/sampling_logp_difference/max": 1.8495888710021973, "sampling/importance_sampling_ratio/min": 0.15730181336402893, "sampling/importance_sampling_ratio/mean": 1.0134751796722412, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9776269495487213, "clip_ratio/low_mean": 0.0936512565240264, "clip_ratio/low_min": 0.0936512565240264, "clip_ratio/high_mean": 0.03307184763252735, "clip_ratio/high_max": 0.03307184763252735, "clip_ratio/region_mean": 0.12672310415655375, "reward_total_mean": 0.5656353831291199, "reward_meter_mean": 0.8966401815414429, "reward_meter_std": 0.14012953639030457, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9699361324310303, "reward_repeat_soft_std": 0.021746700629591942, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.13695022463798523, "reward_total_composite_mean": 0.5656353831291199, "reward_total_composite_std": 0.10508348792791367} {"timestamp_utc": "2026-04-13T10:11:26Z", "mode": "train", "global_step": 1146, "epoch": 0.11511803114013058, "loss": 0.0492, "grad_norm": 15.230727195739746, "learning_rate": 6.530303030303031e-06, "num_tokens": 2027346.0, "completions/mean_length": 36.875, "completions/min_length": 32.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.875, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.9046503305435181, "rewards/meter/std": 0.15936368703842163, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9611223936080933, "rewards/repeat_soft/std": 0.04192990064620972, "rewards/judge_quality/mean": 0.627500057220459, "rewards/judge_quality/std": 0.21952873468399048, "rewards/total_composite/mean": 0.7060563564300537, "rewards/total_composite/std": 0.12517179548740387, "reward": 0.7060563564300537, "reward_std": 0.12517178058624268, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14486348628997803, "sampling/sampling_logp_difference/max": 0.9896438121795654, "sampling/importance_sampling_ratio/min": 0.38649776577949524, "sampling/importance_sampling_ratio/mean": 1.0297596454620361, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0335094183683395, "clip_ratio/low_mean": 0.07998835388571024, "clip_ratio/low_min": 0.07998835388571024, "clip_ratio/high_mean": 0.04699411056935787, "clip_ratio/high_max": 0.04699411056935787, "clip_ratio/region_mean": 0.1269824644550681, "reward_total_mean": 0.7060563564300537, "reward_meter_mean": 0.9046503305435181, "reward_meter_std": 0.15936368703842163, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9611223936080933, "reward_repeat_soft_std": 0.04192990064620972, "reward_judge_quality_mean": 0.627500057220459, "reward_judge_quality_std": 0.21952873468399048, "reward_total_composite_mean": 0.7060563564300537, "reward_total_composite_std": 0.12517179548740387} {"timestamp_utc": "2026-04-13T10:11:33Z", "mode": "train", "global_step": 1147, "epoch": 0.11521848317428428, "loss": 0.0052, "grad_norm": 12.32846736907959, "learning_rate": 6.527272727272728e-06, "num_tokens": 2029195.0, "completions/mean_length": 45.125, "completions/min_length": 41.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.125, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.47214794158935547, "rewards/meter/std": 0.3173607885837555, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.995770275592804, "rewards/repeat_soft/std": 0.0094536691904068, "rewards/judge_quality/mean": 0.7400000095367432, "rewards/judge_quality/std": 0.24859607219696045, "rewards/total_composite/mean": 0.56976318359375, "rewards/total_composite/std": 0.16080376505851746, "reward": 0.56976318359375, "reward_std": 0.16080376505851746, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16579216718673706, "sampling/sampling_logp_difference/max": 2.801460027694702, "sampling/importance_sampling_ratio/min": 0.06072134152054787, "sampling/importance_sampling_ratio/mean": 1.005444049835205, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9021644480526447, "clip_ratio/low_mean": 0.08514065854251385, "clip_ratio/low_min": 0.08514065854251385, "clip_ratio/high_mean": 0.04831978306174278, "clip_ratio/high_max": 0.04831978306174278, "clip_ratio/region_mean": 0.13346044160425663, "reward_total_mean": 0.56976318359375, "reward_meter_mean": 0.47214794158935547, "reward_meter_std": 0.3173607885837555, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.995770275592804, "reward_repeat_soft_std": 0.0094536691904068, "reward_judge_quality_mean": 0.7400000095367432, "reward_judge_quality_std": 0.24859607219696045, "reward_total_composite_mean": 0.56976318359375, "reward_total_composite_std": 0.16080376505851746} {"timestamp_utc": "2026-04-13T10:11:38Z", "mode": "train", "global_step": 1148, "epoch": 0.11531893520843797, "loss": -0.0042, "grad_norm": 12.439154624938965, "learning_rate": 6.524242424242425e-06, "num_tokens": 2030820.0, "completions/mean_length": 48.125, "completions/min_length": 42.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.125, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.7631425261497498, "rewards/meter/std": 0.33935314416885376, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9642658233642578, "rewards/repeat_soft/std": 0.020984716713428497, "rewards/judge_quality/mean": 0.5099999904632568, "rewards/judge_quality/std": 0.13979578018188477, "rewards/total_composite/mean": 0.5847371220588684, "rewards/total_composite/std": 0.10088668018579483, "reward": 0.5847371220588684, "reward_std": 0.10088668018579483, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13191036880016327, "sampling/sampling_logp_difference/max": 1.2770562171936035, "sampling/importance_sampling_ratio/min": 0.2788569927215576, "sampling/importance_sampling_ratio/mean": 1.0432844161987305, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1150656566023827, "clip_ratio/low_mean": 0.03852657042443752, "clip_ratio/low_min": 0.03852657042443752, "clip_ratio/high_mean": 0.08616977697238326, "clip_ratio/high_max": 0.08616977697238326, "clip_ratio/region_mean": 0.12469634739682078, "reward_total_mean": 0.5847371220588684, "reward_meter_mean": 0.7631425261497498, "reward_meter_std": 0.33935314416885376, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9642658233642578, "reward_repeat_soft_std": 0.020984716713428497, "reward_judge_quality_mean": 0.5099999904632568, "reward_judge_quality_std": 0.13979578018188477, "reward_total_composite_mean": 0.5847371220588684, "reward_total_composite_std": 0.10088668018579483} {"timestamp_utc": "2026-04-13T10:11:45Z", "mode": "train", "global_step": 1149, "epoch": 0.11541938724259167, "loss": 0.0663, "grad_norm": 9.508204460144043, "learning_rate": 6.521212121212121e-06, "num_tokens": 2032794.0, "completions/mean_length": 72.75, "completions/min_length": 59.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.75, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.9851264953613281, "rewards/meter/std": 0.006965228822082281, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9444782137870789, "rewards/repeat_soft/std": 0.03553704917430878, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.5871252417564392, "rewards/total_composite/std": 0.12597087025642395, "reward": 0.5871252417564392, "reward_std": 0.12597087025642395, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13400357961654663, "sampling/sampling_logp_difference/max": 2.046328067779541, "sampling/importance_sampling_ratio/min": 0.12920847535133362, "sampling/importance_sampling_ratio/mean": 1.027452826499939, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1493511721491814, "clip_ratio/low_mean": 0.09198104683309793, "clip_ratio/low_min": 0.09198104683309793, "clip_ratio/high_mean": 0.012711863964796066, "clip_ratio/high_max": 0.012711863964796066, "clip_ratio/region_mean": 0.104692910797894, "reward_total_mean": 0.5871252417564392, "reward_meter_mean": 0.9851264953613281, "reward_meter_std": 0.006965228822082281, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9444782137870789, "reward_repeat_soft_std": 0.03553704917430878, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.5871252417564392, "reward_total_composite_std": 0.12597087025642395} {"timestamp_utc": "2026-04-13T10:11:51Z", "mode": "train", "global_step": 1150, "epoch": 0.11551983927674535, "loss": 0.0055, "grad_norm": 9.685317039489746, "learning_rate": 6.5181818181818195e-06, "num_tokens": 2034777.0, "completions/mean_length": 76.875, "completions/min_length": 64.0, "completions/max_length": 89.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.875, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 89.0, "rewards/meter/mean": 0.34443455934524536, "rewards/meter/std": 0.2653238773345947, "rewards/count_adherence/mean": 0.6875, "rewards/count_adherence/std": 0.05892555043101311, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9492671489715576, "rewards/repeat_soft/std": 0.03478346765041351, "rewards/judge_quality/mean": 0.5824999809265137, "rewards/judge_quality/std": 0.176776722073555, "rewards/total_composite/mean": 0.42426055669784546, "rewards/total_composite/std": 0.14111271500587463, "reward": 0.42426055669784546, "reward_std": 0.14111271500587463, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15354497730731964, "sampling/sampling_logp_difference/max": 1.3122196197509766, "sampling/importance_sampling_ratio/min": 0.26922181248664856, "sampling/importance_sampling_ratio/mean": 1.0177932977676392, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.103570282459259, "clip_ratio/low_mean": 0.0882663344964385, "clip_ratio/low_min": 0.0882663344964385, "clip_ratio/high_mean": 0.03345371223986149, "clip_ratio/high_max": 0.03345371223986149, "clip_ratio/region_mean": 0.12172004673629999, "reward_total_mean": 0.42426055669784546, "reward_meter_mean": 0.34443455934524536, "reward_meter_std": 0.2653238773345947, "reward_count_adherence_mean": 0.6875, "reward_count_adherence_std": 0.05892555043101311, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9492671489715576, "reward_repeat_soft_std": 0.03478346765041351, "reward_judge_quality_mean": 0.5824999809265137, "reward_judge_quality_std": 0.176776722073555, "reward_total_composite_mean": 0.42426055669784546, "reward_total_composite_std": 0.14111271500587463} {"timestamp_utc": "2026-04-13T10:12:43Z", "mode": "eval", "global_step": 1150, "epoch": 0.11551983927674535, "eval_loss": NaN, "eval_runtime": 51.4175, "eval_samples_per_second": 1.556, "eval_steps_per_second": 0.194, "eval_num_tokens": 2034777.0, "eval_completions/mean_length": 79.525, "eval_completions/min_length": 30.5, "eval_completions/max_length": 213.0, "eval_completions/clipped_ratio": 0.05, "eval_completions/mean_terminated_length": 56.7702392578125, "eval_completions/min_terminated_length": 30.5, "eval_completions/max_terminated_length": 87.7, "eval_rewards/meter/mean": 0.7342909038066864, "eval_rewards/meter/std": 0.3166229337453842, "eval_rewards/count_adherence/mean": 0.8370833396911621, "eval_rewards/count_adherence/std": 0.1520685613155365, "eval_rewards/hard_gate/mean": 0.95, "eval_rewards/hard_gate/std": 0.11700168251991272, "eval_rewards/repeat_soft/mean": 0.9545246958732605, "eval_rewards/repeat_soft/std": 0.04832870401442051, "eval_rewards/judge_quality/mean": 0.4768750011920929, "eval_rewards/judge_quality/std": 0.19408679232001305, "eval_rewards/total_composite/mean": 0.5226333200931549, "eval_rewards/total_composite/std": 0.16456383764743804, "eval_reward": 0.5226333200931549, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.08284016698598862, "eval_sampling/sampling_logp_difference/max": 0.9862382411956787, "eval_sampling/importance_sampling_ratio/min": 0.37612539529800415, "eval_sampling/importance_sampling_ratio/mean": 1.022798728942871, "eval_sampling/importance_sampling_ratio/max": 1.4848750710487366, "eval_entropy": 0.9602272927761077, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5226333200931549, "eval_reward_meter_mean": 0.7342909038066864, "eval_reward_meter_std": 0.3166229337453842, "eval_reward_count_adherence_mean": 0.8370833396911621, "eval_reward_count_adherence_std": 0.1520685613155365, "eval_reward_hard_gate_mean": 0.95, "eval_reward_hard_gate_std": 0.11700168251991272, "eval_reward_repeat_soft_mean": 0.9545246958732605, "eval_reward_repeat_soft_std": 0.04832870401442051, "eval_reward_judge_quality_mean": 0.4768750011920929, "eval_reward_judge_quality_std": 0.19408679232001305, "eval_reward_total_composite_mean": 0.5226333200931549, "eval_reward_total_composite_std": 0.16456383764743804} {"timestamp_utc": "2026-04-13T10:12:52Z", "mode": "train", "global_step": 1151, "epoch": 0.11562029131089904, "loss": 0.0478, "grad_norm": 11.886629104614258, "learning_rate": 6.515151515151516e-06, "num_tokens": 2036449.0, "completions/mean_length": 52.0, "completions/min_length": 41.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.0, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.8322956562042236, "rewards/meter/std": 0.23459459841251373, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9498786926269531, "rewards/repeat_soft/std": 0.06581722944974899, "rewards/judge_quality/mean": 0.5487500429153442, "rewards/judge_quality/std": 0.22937415540218353, "rewards/total_composite/mean": 0.6352229714393616, "rewards/total_composite/std": 0.14727750420570374, "reward": 0.6352229714393616, "reward_std": 0.14727750420570374, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10419005900621414, "sampling/sampling_logp_difference/max": 1.838627815246582, "sampling/importance_sampling_ratio/min": 0.15903550386428833, "sampling/importance_sampling_ratio/mean": 1.0121780633926392, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3742055706679821, "clip_ratio/low_mean": 0.07162434328347445, "clip_ratio/low_min": 0.07162434328347445, "clip_ratio/high_mean": 0.03747049532830715, "clip_ratio/high_max": 0.03747049532830715, "clip_ratio/region_mean": 0.1090948386117816, "reward_total_mean": 0.6352229714393616, "reward_meter_mean": 0.8322956562042236, "reward_meter_std": 0.23459459841251373, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9498786926269531, "reward_repeat_soft_std": 0.06581722944974899, "reward_judge_quality_mean": 0.5487500429153442, "reward_judge_quality_std": 0.22937415540218353, "reward_total_composite_mean": 0.6352229714393616, "reward_total_composite_std": 0.14727750420570374} {"timestamp_utc": "2026-04-13T10:12:59Z", "mode": "train", "global_step": 1152, "epoch": 0.11572074334505274, "loss": 0.1161, "grad_norm": 15.425283432006836, "learning_rate": 6.512121212121213e-06, "num_tokens": 2037983.0, "completions/mean_length": 41.75, "completions/min_length": 34.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.75, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.8951907753944397, "rewards/meter/std": 0.2683439254760742, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.959107518196106, "rewards/repeat_soft/std": 0.027892453595995903, "rewards/judge_quality/mean": 0.4724999964237213, "rewards/judge_quality/std": 0.11310552060604095, "rewards/total_composite/mean": 0.6202410459518433, "rewards/total_composite/std": 0.11311757564544678, "reward": 0.6202410459518433, "reward_std": 0.11311756819486618, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14502514898777008, "sampling/sampling_logp_difference/max": 1.3253741264343262, "sampling/importance_sampling_ratio/min": 0.26570355892181396, "sampling/importance_sampling_ratio/mean": 1.032871961593628, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8751308396458626, "clip_ratio/low_mean": 0.0844203990418464, "clip_ratio/low_min": 0.0844203990418464, "clip_ratio/high_mean": 0.043231177143752575, "clip_ratio/high_max": 0.043231177143752575, "clip_ratio/region_mean": 0.12765157618559897, "reward_total_mean": 0.6202410459518433, "reward_meter_mean": 0.8951907753944397, "reward_meter_std": 0.2683439254760742, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.959107518196106, "reward_repeat_soft_std": 0.027892453595995903, "reward_judge_quality_mean": 0.4724999964237213, "reward_judge_quality_std": 0.11310552060604095, "reward_total_composite_mean": 0.6202410459518433, "reward_total_composite_std": 0.11311757564544678} {"timestamp_utc": "2026-04-13T10:13:05Z", "mode": "train", "global_step": 1153, "epoch": 0.11582119537920643, "loss": 0.0392, "grad_norm": 15.552083015441895, "learning_rate": 6.5090909090909095e-06, "num_tokens": 2039585.0, "completions/mean_length": 44.25, "completions/min_length": 37.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.25, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.7080739736557007, "rewards/meter/std": 0.23568610846996307, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9552506804466248, "rewards/repeat_soft/std": 0.04192620888352394, "rewards/judge_quality/mean": 0.7100000381469727, "rewards/judge_quality/std": 0.2331768274307251, "rewards/total_composite/mean": 0.6645849943161011, "rewards/total_composite/std": 0.16087202727794647, "reward": 0.6645849943161011, "reward_std": 0.16087201237678528, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11269795894622803, "sampling/sampling_logp_difference/max": 1.5774636268615723, "sampling/importance_sampling_ratio/min": 0.2064981907606125, "sampling/importance_sampling_ratio/mean": 1.0016475915908813, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6939624734222889, "clip_ratio/low_mean": 0.04442748334258795, "clip_ratio/low_min": 0.04442748334258795, "clip_ratio/high_mean": 0.06681038066744804, "clip_ratio/high_max": 0.06681038066744804, "clip_ratio/region_mean": 0.11123786401003599, "reward_total_mean": 0.6645849943161011, "reward_meter_mean": 0.7080739736557007, "reward_meter_std": 0.23568610846996307, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9552506804466248, "reward_repeat_soft_std": 0.04192620888352394, "reward_judge_quality_mean": 0.7100000381469727, "reward_judge_quality_std": 0.2331768274307251, "reward_total_composite_mean": 0.6645849943161011, "reward_total_composite_std": 0.16087202727794647} {"timestamp_utc": "2026-04-13T10:13:17Z", "mode": "train", "global_step": 1154, "epoch": 0.11592164741336013, "loss": -0.1162, "grad_norm": 2.4770925045013428, "learning_rate": 6.506060606060607e-06, "num_tokens": 2041124.0, "completions/mean_length": 98.375, "completions/min_length": 37.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 39.28571701049805, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.9925278425216675, "rewards/meter/std": 0.003449532203376293, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9869427680969238, "rewards/repeat_soft/std": 0.02140633761882782, "rewards/judge_quality/mean": 0.5087500214576721, "rewards/judge_quality/std": 0.24002604186534882, "rewards/total_composite/mean": 0.6286358833312988, "rewards/total_composite/std": 0.2727552056312561, "reward": 0.6286358833312988, "reward_std": 0.2727552056312561, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14411436021327972, "sampling/sampling_logp_difference/max": 1.1456871032714844, "sampling/importance_sampling_ratio/min": 0.3180053234100342, "sampling/importance_sampling_ratio/mean": 1.0198239088058472, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8729628100991249, "clip_ratio/low_mean": 0.03125, "clip_ratio/low_min": 0.03125, "clip_ratio/high_mean": 0.0861591063439846, "clip_ratio/high_max": 0.0861591063439846, "clip_ratio/region_mean": 0.1174091063439846, "reward_total_mean": 0.6286358833312988, "reward_meter_mean": 0.9925278425216675, "reward_meter_std": 0.003449532203376293, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9869427680969238, "reward_repeat_soft_std": 0.02140633761882782, "reward_judge_quality_mean": 0.5087500214576721, "reward_judge_quality_std": 0.24002604186534882, "reward_total_composite_mean": 0.6286358833312988, "reward_total_composite_std": 0.2727552056312561} {"timestamp_utc": "2026-04-13T10:13:23Z", "mode": "train", "global_step": 1155, "epoch": 0.11602209944751381, "loss": 0.0446, "grad_norm": 13.718782424926758, "learning_rate": 6.503030303030303e-06, "num_tokens": 2042771.0, "completions/mean_length": 46.875, "completions/min_length": 36.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.875, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.8706246614456177, "rewards/meter/std": 0.25902047753334045, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9424266815185547, "rewards/repeat_soft/std": 0.040095433592796326, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.10392305999994278, "rewards/total_composite/mean": 0.5893842577934265, "rewards/total_composite/std": 0.05839797854423523, "reward": 0.5893842577934265, "reward_std": 0.058397967368364334, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.143097922205925, "sampling/sampling_logp_difference/max": 1.1727938652038574, "sampling/importance_sampling_ratio/min": 0.3095010221004486, "sampling/importance_sampling_ratio/mean": 1.0094670057296753, "sampling/importance_sampling_ratio/max": 1.9955836534500122, "entropy": 1.087383322417736, "clip_ratio/low_mean": 0.022058824077248573, "clip_ratio/low_min": 0.022058824077248573, "clip_ratio/high_mean": 0.10677625681273639, "clip_ratio/high_max": 0.10677625681273639, "clip_ratio/region_mean": 0.12883508088998497, "reward_total_mean": 0.5893842577934265, "reward_meter_mean": 0.8706246614456177, "reward_meter_std": 0.25902047753334045, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9424266815185547, "reward_repeat_soft_std": 0.040095433592796326, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.10392305999994278, "reward_total_composite_mean": 0.5893842577934265, "reward_total_composite_std": 0.05839797854423523} {"timestamp_utc": "2026-04-13T10:13:31Z", "mode": "train", "global_step": 1156, "epoch": 0.1161225514816675, "loss": -0.0461, "grad_norm": 12.073512077331543, "learning_rate": 6.5000000000000004e-06, "num_tokens": 2044454.0, "completions/mean_length": 41.375, "completions/min_length": 36.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.375, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.9720487594604492, "rewards/meter/std": 0.018122032284736633, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9285119771957397, "rewards/repeat_soft/std": 0.011059910990297794, "rewards/judge_quality/mean": 0.5049999952316284, "rewards/judge_quality/std": 0.16801361739635468, "rewards/total_composite/mean": 0.6586222648620605, "rewards/total_composite/std": 0.10675440728664398, "reward": 0.6586222648620605, "reward_std": 0.10675440728664398, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1238454058766365, "sampling/sampling_logp_difference/max": 2.65266752243042, "sampling/importance_sampling_ratio/min": 0.07046300172805786, "sampling/importance_sampling_ratio/mean": 1.0210554599761963, "sampling/importance_sampling_ratio/max": 1.9586304426193237, "entropy": 1.0119702145457268, "clip_ratio/low_mean": 0.10703061474487185, "clip_ratio/low_min": 0.10703061474487185, "clip_ratio/high_mean": 0.018617020919919014, "clip_ratio/high_max": 0.018617020919919014, "clip_ratio/region_mean": 0.12564763566479087, "reward_total_mean": 0.6586222648620605, "reward_meter_mean": 0.9720487594604492, "reward_meter_std": 0.018122032284736633, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9285119771957397, "reward_repeat_soft_std": 0.011059910990297794, "reward_judge_quality_mean": 0.5049999952316284, "reward_judge_quality_std": 0.16801361739635468, "reward_total_composite_mean": 0.6586222648620605, "reward_total_composite_std": 0.10675440728664398} {"timestamp_utc": "2026-04-13T10:13:37Z", "mode": "train", "global_step": 1157, "epoch": 0.1162230035158212, "loss": -0.022, "grad_norm": 10.855276107788086, "learning_rate": 6.496969696969697e-06, "num_tokens": 2046156.0, "completions/mean_length": 43.75, "completions/min_length": 38.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.75, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.8936759233474731, "rewards/meter/std": 0.1595841944217682, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.895618736743927, "rewards/repeat_soft/std": 0.0682944655418396, "rewards/judge_quality/mean": 0.5637500286102295, "rewards/judge_quality/std": 0.22012579441070557, "rewards/total_composite/mean": 0.6663017868995667, "rewards/total_composite/std": 0.15666808187961578, "reward": 0.6663017868995667, "reward_std": 0.1566680669784546, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11408718675374985, "sampling/sampling_logp_difference/max": 1.7818901538848877, "sampling/importance_sampling_ratio/min": 0.27114689350128174, "sampling/importance_sampling_ratio/mean": 1.0291129350662231, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.74884133040905, "clip_ratio/low_mean": 0.0684937359765172, "clip_ratio/low_min": 0.0684937359765172, "clip_ratio/high_mean": 0.03380434773862362, "clip_ratio/high_max": 0.03380434773862362, "clip_ratio/region_mean": 0.10229808371514082, "reward_total_mean": 0.6663017868995667, "reward_meter_mean": 0.8936759233474731, "reward_meter_std": 0.1595841944217682, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.895618736743927, "reward_repeat_soft_std": 0.0682944655418396, "reward_judge_quality_mean": 0.5637500286102295, "reward_judge_quality_std": 0.22012579441070557, "reward_total_composite_mean": 0.6663017868995667, "reward_total_composite_std": 0.15666808187961578} {"timestamp_utc": "2026-04-13T10:13:48Z", "mode": "train", "global_step": 1158, "epoch": 0.11632345554997489, "loss": -0.1513, "grad_norm": 3.5333056449890137, "learning_rate": 6.493939393939395e-06, "num_tokens": 2048145.0, "completions/mean_length": 124.625, "completions/min_length": 52.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 69.28572082519531, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.7109284996986389, "rewards/meter/std": 0.3987520635128021, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9296675324440002, "rewards/repeat_soft/std": 0.04415125399827957, "rewards/judge_quality/mean": 0.4649999737739563, "rewards/judge_quality/std": 0.3147788643836975, "rewards/total_composite/mean": 0.5115648508071899, "rewards/total_composite/std": 0.2610779404640198, "reward": 0.5115648508071899, "reward_std": 0.2610779404640198, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12852534651756287, "sampling/sampling_logp_difference/max": 1.8264281749725342, "sampling/importance_sampling_ratio/min": 0.16098757088184357, "sampling/importance_sampling_ratio/mean": 1.0188608169555664, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7715061381459236, "clip_ratio/low_mean": 0.03934571333229542, "clip_ratio/low_min": 0.03934571333229542, "clip_ratio/high_mean": 0.08331483043730259, "clip_ratio/high_max": 0.08331483043730259, "clip_ratio/region_mean": 0.12266054376959801, "reward_total_mean": 0.5115648508071899, "reward_meter_mean": 0.7109284996986389, "reward_meter_std": 0.3987520635128021, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9296675324440002, "reward_repeat_soft_std": 0.04415125399827957, "reward_judge_quality_mean": 0.4649999737739563, "reward_judge_quality_std": 0.3147788643836975, "reward_total_composite_mean": 0.5115648508071899, "reward_total_composite_std": 0.2610779404640198} {"timestamp_utc": "2026-04-13T10:13:54Z", "mode": "train", "global_step": 1159, "epoch": 0.11642390758412857, "loss": 0.0341, "grad_norm": 16.398767471313477, "learning_rate": 6.490909090909091e-06, "num_tokens": 2049736.0, "completions/mean_length": 39.875, "completions/min_length": 35.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.875, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.9786190390586853, "rewards/meter/std": 0.013541036285459995, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9787521362304688, "rewards/repeat_soft/std": 0.027974126860499382, "rewards/judge_quality/mean": 0.7200000286102295, "rewards/judge_quality/std": 0.20701968669891357, "rewards/total_composite/mean": 0.8045221567153931, "rewards/total_composite/std": 0.13010771572589874, "reward": 0.8045221567153931, "reward_std": 0.13010771572589874, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1522352546453476, "sampling/sampling_logp_difference/max": 1.1938486099243164, "sampling/importance_sampling_ratio/min": 0.303052693605423, "sampling/importance_sampling_ratio/mean": 1.0002986192703247, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0424583405256271, "clip_ratio/low_mean": 0.04302912671118975, "clip_ratio/low_min": 0.04302912671118975, "clip_ratio/high_mean": 0.09753315430134535, "clip_ratio/high_max": 0.09753315430134535, "clip_ratio/region_mean": 0.1405622810125351, "reward_total_mean": 0.8045221567153931, "reward_meter_mean": 0.9786190390586853, "reward_meter_std": 0.013541036285459995, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9787521362304688, "reward_repeat_soft_std": 0.027974126860499382, "reward_judge_quality_mean": 0.7200000286102295, "reward_judge_quality_std": 0.20701968669891357, "reward_total_composite_mean": 0.8045221567153931, "reward_total_composite_std": 0.13010771572589874} {"timestamp_utc": "2026-04-13T10:14:01Z", "mode": "train", "global_step": 1160, "epoch": 0.11652435961828227, "loss": -0.0229, "grad_norm": 16.45473861694336, "learning_rate": 6.487878787878789e-06, "num_tokens": 2050986.0, "completions/mean_length": 19.25, "completions/min_length": 16.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 19.25, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.33942797780036926, "rewards/meter/std": 0.3242776095867157, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9583057761192322, "rewards/repeat_soft/std": 0.007864047773182392, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.3720680773258209, "rewards/total_composite/std": 0.17117416858673096, "reward": 0.3720680773258209, "reward_std": 0.17117416858673096, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15709532797336578, "sampling/sampling_logp_difference/max": 1.241135597229004, "sampling/importance_sampling_ratio/min": 0.2890557646751404, "sampling/importance_sampling_ratio/mean": 1.0157854557037354, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1322322264313698, "clip_ratio/low_mean": 0.06802545534446836, "clip_ratio/low_min": 0.06802545534446836, "clip_ratio/high_mean": 0.09996639750897884, "clip_ratio/high_max": 0.09996639750897884, "clip_ratio/region_mean": 0.1679918528534472, "reward_total_mean": 0.3720680773258209, "reward_meter_mean": 0.33942797780036926, "reward_meter_std": 0.3242776095867157, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9583057761192322, "reward_repeat_soft_std": 0.007864047773182392, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.3720680773258209, "reward_total_composite_std": 0.17117416858673096} {"timestamp_utc": "2026-04-13T10:14:07Z", "mode": "train", "global_step": 1161, "epoch": 0.11662481165243596, "loss": 0.0639, "grad_norm": 19.172563552856445, "learning_rate": 6.484848484848485e-06, "num_tokens": 2052447.0, "completions/mean_length": 18.625, "completions/min_length": 15.0, "completions/max_length": 23.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 18.625, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 23.0, "rewards/meter/mean": 0.7257708311080933, "rewards/meter/std": 0.25202879309654236, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9509689807891846, "rewards/repeat_soft/std": 0.02166803367435932, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.540780782699585, "rewards/total_composite/std": 0.06680676341056824, "reward": 0.540780782699585, "reward_std": 0.06680675595998764, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18771231174468994, "sampling/sampling_logp_difference/max": 1.4687366485595703, "sampling/importance_sampling_ratio/min": 0.2302161604166031, "sampling/importance_sampling_ratio/mean": 1.011730670928955, "sampling/importance_sampling_ratio/max": 1.923264980316162, "entropy": 1.036086767911911, "clip_ratio/low_mean": 0.11173768062144518, "clip_ratio/low_min": 0.11173768062144518, "clip_ratio/high_mean": 0.09665388148277998, "clip_ratio/high_max": 0.09665388148277998, "clip_ratio/region_mean": 0.20839156210422516, "reward_total_mean": 0.540780782699585, "reward_meter_mean": 0.7257708311080933, "reward_meter_std": 0.25202879309654236, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9509689807891846, "reward_repeat_soft_std": 0.02166803367435932, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.540780782699585, "reward_total_composite_std": 0.06680676341056824} {"timestamp_utc": "2026-04-13T10:14:14Z", "mode": "train", "global_step": 1162, "epoch": 0.11672526368658966, "loss": -0.0015, "grad_norm": 9.042016983032227, "learning_rate": 6.481818181818182e-06, "num_tokens": 2054554.0, "completions/mean_length": 80.375, "completions/min_length": 69.0, "completions/max_length": 100.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.375, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 100.0, "rewards/meter/mean": 0.9644762277603149, "rewards/meter/std": 0.05867907777428627, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.938398003578186, "rewards/repeat_soft/std": 0.02318570762872696, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5540616512298584, "rewards/total_composite/std": 0.01620873436331749, "reward": 0.5540616512298584, "reward_std": 0.01620873622596264, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14428913593292236, "sampling/sampling_logp_difference/max": 2.3007876873016357, "sampling/importance_sampling_ratio/min": 0.10017991065979004, "sampling/importance_sampling_ratio/mean": 1.0051047801971436, "sampling/importance_sampling_ratio/max": 1.9472465515136719, "entropy": 0.9574398621916771, "clip_ratio/low_mean": 0.039586388505995274, "clip_ratio/low_min": 0.039586388505995274, "clip_ratio/high_mean": 0.07957096491008997, "clip_ratio/high_max": 0.07957096491008997, "clip_ratio/region_mean": 0.11915735341608524, "reward_total_mean": 0.5540616512298584, "reward_meter_mean": 0.9644762277603149, "reward_meter_std": 0.05867907777428627, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.938398003578186, "reward_repeat_soft_std": 0.02318570762872696, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5540616512298584, "reward_total_composite_std": 0.01620873436331749} {"timestamp_utc": "2026-04-13T10:14:20Z", "mode": "train", "global_step": 1163, "epoch": 0.11682571572074335, "loss": 0.0314, "grad_norm": 16.131433486938477, "learning_rate": 6.478787878787879e-06, "num_tokens": 2056378.0, "completions/mean_length": 52.0, "completions/min_length": 49.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.0, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.7054102420806885, "rewards/meter/std": 0.28646501898765564, "rewards/count_adherence/mean": 0.7250000238418579, "rewards/count_adherence/std": 0.1035098284482956, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9489527344703674, "rewards/repeat_soft/std": 0.03885876014828682, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.45700687170028687, "rewards/total_composite/std": 0.10430697351694107, "reward": 0.45700687170028687, "reward_std": 0.10430697351694107, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13044683635234833, "sampling/sampling_logp_difference/max": 2.052244186401367, "sampling/importance_sampling_ratio/min": 0.1284463107585907, "sampling/importance_sampling_ratio/mean": 0.9911546111106873, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6575556993484497, "clip_ratio/low_mean": 0.05782278999686241, "clip_ratio/low_min": 0.05782278999686241, "clip_ratio/high_mean": 0.04991029482334852, "clip_ratio/high_max": 0.04991029482334852, "clip_ratio/region_mean": 0.10773308482021093, "reward_total_mean": 0.45700687170028687, "reward_meter_mean": 0.7054102420806885, "reward_meter_std": 0.28646501898765564, "reward_count_adherence_mean": 0.7250000238418579, "reward_count_adherence_std": 0.1035098284482956, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9489527344703674, "reward_repeat_soft_std": 0.03885876014828682, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.45700687170028687, "reward_total_composite_std": 0.10430697351694107} {"timestamp_utc": "2026-04-13T10:14:26Z", "mode": "train", "global_step": 1164, "epoch": 0.11692616775489703, "loss": -0.0325, "grad_norm": 19.62907600402832, "learning_rate": 6.475757575757576e-06, "num_tokens": 2057812.0, "completions/mean_length": 20.25, "completions/min_length": 17.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 20.25, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9464387893676758, "rewards/meter/std": 0.09658417850732803, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.961837112903595, "rewards/repeat_soft/std": 0.0018749026348814368, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.6116173267364502, "rewards/total_composite/std": 0.0269559845328331, "reward": 0.6116173267364502, "reward_std": 0.026955988258123398, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15353576838970184, "sampling/sampling_logp_difference/max": 1.5081539154052734, "sampling/importance_sampling_ratio/min": 0.22131817042827606, "sampling/importance_sampling_ratio/mean": 0.997308075428009, "sampling/importance_sampling_ratio/max": 1.7216521501541138, "entropy": 1.0296223536133766, "clip_ratio/low_mean": 0.035153554286807775, "clip_ratio/low_min": 0.035153554286807775, "clip_ratio/high_mean": 0.06448688125237823, "clip_ratio/high_max": 0.06448688125237823, "clip_ratio/region_mean": 0.099640435539186, "reward_total_mean": 0.6116173267364502, "reward_meter_mean": 0.9464387893676758, "reward_meter_std": 0.09658417850732803, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.961837112903595, "reward_repeat_soft_std": 0.0018749026348814368, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.6116173267364502, "reward_total_composite_std": 0.0269559845328331} {"timestamp_utc": "2026-04-13T10:14:33Z", "mode": "train", "global_step": 1165, "epoch": 0.11702661978905073, "loss": 0.0137, "grad_norm": 8.567177772521973, "learning_rate": 6.472727272727272e-06, "num_tokens": 2060085.0, "completions/mean_length": 103.125, "completions/min_length": 95.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 103.125, "completions/min_terminated_length": 95.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.9605042338371277, "rewards/meter/std": 0.04659838229417801, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9363154768943787, "rewards/repeat_soft/std": 0.03352692350745201, "rewards/judge_quality/mean": 0.5399999618530273, "rewards/judge_quality/std": 0.22245386242866516, "rewards/total_composite/mean": 0.4727051556110382, "rewards/total_composite/std": 0.3168168067932129, "reward": 0.4727051556110382, "reward_std": 0.3168168067932129, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14320671558380127, "sampling/sampling_logp_difference/max": 1.9507179260253906, "sampling/importance_sampling_ratio/min": 0.1421719640493393, "sampling/importance_sampling_ratio/mean": 1.007926344871521, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0821733325719833, "clip_ratio/low_mean": 0.0317208468914032, "clip_ratio/low_min": 0.0317208468914032, "clip_ratio/high_mean": 0.09785328712314367, "clip_ratio/high_max": 0.09785328712314367, "clip_ratio/region_mean": 0.12957413401454687, "reward_total_mean": 0.4727051556110382, "reward_meter_mean": 0.9605042338371277, "reward_meter_std": 0.04659838229417801, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9363154768943787, "reward_repeat_soft_std": 0.03352692350745201, "reward_judge_quality_mean": 0.5399999618530273, "reward_judge_quality_std": 0.22245386242866516, "reward_total_composite_mean": 0.4727051556110382, "reward_total_composite_std": 0.3168168067932129} {"timestamp_utc": "2026-04-13T10:14:39Z", "mode": "train", "global_step": 1166, "epoch": 0.11712707182320442, "loss": 0.0913, "grad_norm": 9.98704719543457, "learning_rate": 6.4696969696969705e-06, "num_tokens": 2062101.0, "completions/mean_length": 76.0, "completions/min_length": 62.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.0, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.9051650166511536, "rewards/meter/std": 0.17993085086345673, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9518389701843262, "rewards/repeat_soft/std": 0.056906893849372864, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5898858904838562, "rewards/total_composite/std": 0.04827408120036125, "reward": 0.5898858904838562, "reward_std": 0.048274096101522446, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13661645352840424, "sampling/sampling_logp_difference/max": 2.071254014968872, "sampling/importance_sampling_ratio/min": 0.1260276436805725, "sampling/importance_sampling_ratio/mean": 1.0345901250839233, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0170091539621353, "clip_ratio/low_mean": 0.05436989106237888, "clip_ratio/low_min": 0.05436989106237888, "clip_ratio/high_mean": 0.07861310988664627, "clip_ratio/high_max": 0.07861310988664627, "clip_ratio/region_mean": 0.13298300094902515, "reward_total_mean": 0.5898858904838562, "reward_meter_mean": 0.9051650166511536, "reward_meter_std": 0.17993085086345673, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9518389701843262, "reward_repeat_soft_std": 0.056906893849372864, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5898858904838562, "reward_total_composite_std": 0.04827408120036125} {"timestamp_utc": "2026-04-13T10:14:46Z", "mode": "train", "global_step": 1167, "epoch": 0.11722752385735812, "loss": 0.0675, "grad_norm": 18.763057708740234, "learning_rate": 6.466666666666667e-06, "num_tokens": 2063765.0, "completions/mean_length": 35.0, "completions/min_length": 27.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.0, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.3749932050704956, "rewards/meter/std": 0.3746505677700043, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9954110383987427, "rewards/repeat_soft/std": 0.00569864921271801, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.4894949495792389, "rewards/total_composite/std": 0.1945095956325531, "reward": 0.4894949495792389, "reward_std": 0.1945095956325531, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1903914213180542, "sampling/sampling_logp_difference/max": 2.062476634979248, "sampling/importance_sampling_ratio/min": 0.12713870406150818, "sampling/importance_sampling_ratio/mean": 1.0226589441299438, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8588284626603127, "clip_ratio/low_mean": 0.07798821665346622, "clip_ratio/low_min": 0.07798821665346622, "clip_ratio/high_mean": 0.06570973061025143, "clip_ratio/high_max": 0.06570973061025143, "clip_ratio/region_mean": 0.14369794726371765, "reward_total_mean": 0.4894949495792389, "reward_meter_mean": 0.3749932050704956, "reward_meter_std": 0.3746505677700043, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9954110383987427, "reward_repeat_soft_std": 0.00569864921271801, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.4894949495792389, "reward_total_composite_std": 0.1945095956325531} {"timestamp_utc": "2026-04-13T10:14:52Z", "mode": "train", "global_step": 1168, "epoch": 0.11732797589151181, "loss": 0.0471, "grad_norm": 15.621354103088379, "learning_rate": 6.463636363636364e-06, "num_tokens": 2065411.0, "completions/mean_length": 35.75, "completions/min_length": 32.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.75, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.7622805833816528, "rewards/meter/std": 0.3560439348220825, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.896129310131073, "rewards/repeat_soft/std": 0.07596825063228607, "rewards/judge_quality/mean": 0.5862500071525574, "rewards/judge_quality/std": 0.2353682667016983, "rewards/total_composite/mean": 0.6478046178817749, "rewards/total_composite/std": 0.21685576438903809, "reward": 0.6478046178817749, "reward_std": 0.21685576438903809, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13290895521640778, "sampling/sampling_logp_difference/max": 1.4618310928344727, "sampling/importance_sampling_ratio/min": 0.23181141912937164, "sampling/importance_sampling_ratio/mean": 1.0047551393508911, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6507822498679161, "clip_ratio/low_mean": 0.06357021653093398, "clip_ratio/low_min": 0.06357021653093398, "clip_ratio/high_mean": 0.06838235631585121, "clip_ratio/high_max": 0.06838235631585121, "clip_ratio/region_mean": 0.1319525728467852, "reward_total_mean": 0.6478046178817749, "reward_meter_mean": 0.7622805833816528, "reward_meter_std": 0.3560439348220825, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.896129310131073, "reward_repeat_soft_std": 0.07596825063228607, "reward_judge_quality_mean": 0.5862500071525574, "reward_judge_quality_std": 0.2353682667016983, "reward_total_composite_mean": 0.6478046178817749, "reward_total_composite_std": 0.21685576438903809} {"timestamp_utc": "2026-04-13T10:14:58Z", "mode": "train", "global_step": 1169, "epoch": 0.11742842792566549, "loss": 0.0867, "grad_norm": 14.500654220581055, "learning_rate": 6.460606060606061e-06, "num_tokens": 2067120.0, "completions/mean_length": 50.625, "completions/min_length": 39.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.625, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.791524350643158, "rewards/meter/std": 0.24436300992965698, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9756710529327393, "rewards/repeat_soft/std": 0.03941075876355171, "rewards/judge_quality/mean": 0.5274999737739563, "rewards/judge_quality/std": 0.18873640894889832, "rewards/total_composite/mean": 0.6255395412445068, "rewards/total_composite/std": 0.14700382947921753, "reward": 0.6255395412445068, "reward_std": 0.14700381457805634, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1730264574289322, "sampling/sampling_logp_difference/max": 1.6852831840515137, "sampling/importance_sampling_ratio/min": 0.18539191782474518, "sampling/importance_sampling_ratio/mean": 1.0151746273040771, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2209726721048355, "clip_ratio/low_mean": 0.07802170515060425, "clip_ratio/low_min": 0.07802170515060425, "clip_ratio/high_mean": 0.059146737679839134, "clip_ratio/high_max": 0.059146737679839134, "clip_ratio/region_mean": 0.13716844283044338, "reward_total_mean": 0.6255395412445068, "reward_meter_mean": 0.791524350643158, "reward_meter_std": 0.24436300992965698, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9756710529327393, "reward_repeat_soft_std": 0.03941075876355171, "reward_judge_quality_mean": 0.5274999737739563, "reward_judge_quality_std": 0.18873640894889832, "reward_total_composite_mean": 0.6255395412445068, "reward_total_composite_std": 0.14700382947921753} {"timestamp_utc": "2026-04-13T10:15:05Z", "mode": "train", "global_step": 1170, "epoch": 0.11752887995981919, "loss": 0.0595, "grad_norm": 13.179718017578125, "learning_rate": 6.457575757575758e-06, "num_tokens": 2068715.0, "completions/mean_length": 43.375, "completions/min_length": 35.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.375, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.8890571594238281, "rewards/meter/std": 0.28032368421554565, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9450587034225464, "rewards/repeat_soft/std": 0.10344314575195312, "rewards/judge_quality/mean": 0.3474999964237213, "rewards/judge_quality/std": 0.11310551315546036, "rewards/total_composite/mean": 0.5488234162330627, "rewards/total_composite/std": 0.0971178188920021, "reward": 0.5488234162330627, "reward_std": 0.09711781144142151, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12511882185935974, "sampling/sampling_logp_difference/max": 1.342888355255127, "sampling/importance_sampling_ratio/min": 0.2610904574394226, "sampling/importance_sampling_ratio/mean": 1.0211719274520874, "sampling/importance_sampling_ratio/max": 1.907301664352417, "entropy": 1.0861545503139496, "clip_ratio/low_mean": 0.05386972241103649, "clip_ratio/low_min": 0.05386972241103649, "clip_ratio/high_mean": 0.09366679657250643, "clip_ratio/high_max": 0.09366679657250643, "clip_ratio/region_mean": 0.14753651898354292, "reward_total_mean": 0.5488234162330627, "reward_meter_mean": 0.8890571594238281, "reward_meter_std": 0.28032368421554565, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9450587034225464, "reward_repeat_soft_std": 0.10344314575195312, "reward_judge_quality_mean": 0.3474999964237213, "reward_judge_quality_std": 0.11310551315546036, "reward_total_composite_mean": 0.5488234162330627, "reward_total_composite_std": 0.0971178188920021} {"timestamp_utc": "2026-04-13T10:15:16Z", "mode": "train", "global_step": 1171, "epoch": 0.11762933199397288, "loss": -0.1137, "grad_norm": 2.412520408630371, "learning_rate": 6.454545454545456e-06, "num_tokens": 2070387.0, "completions/mean_length": 106.0, "completions/min_length": 42.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 48.000003814697266, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.8596898317337036, "rewards/meter/std": 0.2784144878387451, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9502856135368347, "rewards/repeat_soft/std": 0.050328876823186874, "rewards/judge_quality/mean": 0.5062500238418579, "rewards/judge_quality/std": 0.2876474857330322, "rewards/total_composite/mean": 0.6112303137779236, "rewards/total_composite/std": 0.28258681297302246, "reward": 0.6112303137779236, "reward_std": 0.28258681297302246, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13302764296531677, "sampling/sampling_logp_difference/max": 4.004399299621582, "sampling/importance_sampling_ratio/min": 0.018235240131616592, "sampling/importance_sampling_ratio/mean": 1.0074357986450195, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7949735596776009, "clip_ratio/low_mean": 0.034433962777256966, "clip_ratio/low_min": 0.034433962777256966, "clip_ratio/high_mean": 0.06723435246385634, "clip_ratio/high_max": 0.06723435246385634, "clip_ratio/region_mean": 0.1016683152411133, "reward_total_mean": 0.6112303137779236, "reward_meter_mean": 0.8596898317337036, "reward_meter_std": 0.2784144878387451, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9502856135368347, "reward_repeat_soft_std": 0.050328876823186874, "reward_judge_quality_mean": 0.5062500238418579, "reward_judge_quality_std": 0.2876474857330322, "reward_total_composite_mean": 0.6112303137779236, "reward_total_composite_std": 0.28258681297302246} {"timestamp_utc": "2026-04-13T10:15:23Z", "mode": "train", "global_step": 1172, "epoch": 0.11772978402812657, "loss": 0.0459, "grad_norm": 9.369461059570312, "learning_rate": 6.451515151515152e-06, "num_tokens": 2072506.0, "completions/mean_length": 84.875, "completions/min_length": 67.0, "completions/max_length": 97.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.875, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.9398049712181091, "rewards/meter/std": 0.09466666728258133, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8908748626708984, "rewards/repeat_soft/std": 0.05745447054505348, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.517151415348053, "rewards/total_composite/std": 0.06308780610561371, "reward": 0.517151415348053, "reward_std": 0.06308779865503311, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14235681295394897, "sampling/sampling_logp_difference/max": 1.635481834411621, "sampling/importance_sampling_ratio/min": 0.19485846161842346, "sampling/importance_sampling_ratio/mean": 1.026593565940857, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.134580135345459, "clip_ratio/low_mean": 0.03076496720314026, "clip_ratio/low_min": 0.03076496720314026, "clip_ratio/high_mean": 0.10033840034157038, "clip_ratio/high_max": 0.10033840034157038, "clip_ratio/region_mean": 0.13110336754471064, "reward_total_mean": 0.517151415348053, "reward_meter_mean": 0.9398049712181091, "reward_meter_std": 0.09466666728258133, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8908748626708984, "reward_repeat_soft_std": 0.05745447054505348, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.517151415348053, "reward_total_composite_std": 0.06308780610561371} {"timestamp_utc": "2026-04-13T10:15:30Z", "mode": "train", "global_step": 1173, "epoch": 0.11783023606228026, "loss": -0.031, "grad_norm": 13.581130981445312, "learning_rate": 6.4484848484848496e-06, "num_tokens": 2074062.0, "completions/mean_length": 30.5, "completions/min_length": 24.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.5, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.8466594815254211, "rewards/meter/std": 0.31667131185531616, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9836012125015259, "rewards/repeat_soft/std": 0.007643247488886118, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.6549550294876099, "rewards/total_composite/std": 0.17814242839813232, "reward": 0.6549550294876099, "reward_std": 0.17814242839813232, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11450258642435074, "sampling/sampling_logp_difference/max": 1.3845956325531006, "sampling/importance_sampling_ratio/min": 0.2504250407218933, "sampling/importance_sampling_ratio/mean": 0.9979570508003235, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6376455649733543, "clip_ratio/low_mean": 0.08072647964581847, "clip_ratio/low_min": 0.08072647964581847, "clip_ratio/high_mean": 0.026751894503831863, "clip_ratio/high_max": 0.026751894503831863, "clip_ratio/region_mean": 0.10747837414965034, "reward_total_mean": 0.6549550294876099, "reward_meter_mean": 0.8466594815254211, "reward_meter_std": 0.31667131185531616, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9836012125015259, "reward_repeat_soft_std": 0.007643247488886118, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.6549550294876099, "reward_total_composite_std": 0.17814242839813232} {"timestamp_utc": "2026-04-13T10:15:37Z", "mode": "train", "global_step": 1174, "epoch": 0.11793068809643395, "loss": 0.0277, "grad_norm": 15.254562377929688, "learning_rate": 6.445454545454546e-06, "num_tokens": 2075722.0, "completions/mean_length": 43.5, "completions/min_length": 38.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.5, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.7098647356033325, "rewards/meter/std": 0.26970329880714417, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9789924621582031, "rewards/repeat_soft/std": 0.03237193077802658, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5406419038772583, "rewards/total_composite/std": 0.07226701825857162, "reward": 0.5406419038772583, "reward_std": 0.07226699590682983, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16433273255825043, "sampling/sampling_logp_difference/max": 2.1549975872039795, "sampling/importance_sampling_ratio/min": 0.11590346693992615, "sampling/importance_sampling_ratio/mean": 1.0102665424346924, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.253235936164856, "clip_ratio/low_mean": 0.044484478421509266, "clip_ratio/low_min": 0.044484478421509266, "clip_ratio/high_mean": 0.10984905622899532, "clip_ratio/high_max": 0.10984905622899532, "clip_ratio/region_mean": 0.1543335346505046, "reward_total_mean": 0.5406419038772583, "reward_meter_mean": 0.7098647356033325, "reward_meter_std": 0.26970329880714417, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9789924621582031, "reward_repeat_soft_std": 0.03237193077802658, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5406419038772583, "reward_total_composite_std": 0.07226701825857162} {"timestamp_utc": "2026-04-13T10:15:44Z", "mode": "train", "global_step": 1175, "epoch": 0.11803114013058764, "loss": 0.0175, "grad_norm": 10.267904281616211, "learning_rate": 6.442424242424243e-06, "num_tokens": 2078073.0, "completions/mean_length": 75.875, "completions/min_length": 65.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.875, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.9792505502700806, "rewards/meter/std": 0.010109245777130127, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9222643971443176, "rewards/repeat_soft/std": 0.050244636833667755, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.539008378982544, "rewards/total_composite/std": 0.007422552909702063, "reward": 0.539008378982544, "reward_std": 0.007422561291605234, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1581365019083023, "sampling/sampling_logp_difference/max": 2.3960189819335938, "sampling/importance_sampling_ratio/min": 0.09107982367277145, "sampling/importance_sampling_ratio/mean": 1.0091606378555298, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7431683018803596, "clip_ratio/low_mean": 0.09852162003517151, "clip_ratio/low_min": 0.09852162003517151, "clip_ratio/high_mean": 0.04452393390238285, "clip_ratio/high_max": 0.04452393390238285, "clip_ratio/region_mean": 0.14304555393755436, "reward_total_mean": 0.539008378982544, "reward_meter_mean": 0.9792505502700806, "reward_meter_std": 0.010109245777130127, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9222643971443176, "reward_repeat_soft_std": 0.050244636833667755, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.539008378982544, "reward_total_composite_std": 0.007422552909702063} {"timestamp_utc": "2026-04-13T10:15:51Z", "mode": "train", "global_step": 1176, "epoch": 0.11813159216474134, "loss": 0.0091, "grad_norm": 12.990802764892578, "learning_rate": 6.43939393939394e-06, "num_tokens": 2079566.0, "completions/mean_length": 27.625, "completions/min_length": 26.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.625, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.7428567409515381, "rewards/meter/std": 0.3768162429332733, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.8025000095367432, "rewards/judge_quality/std": 0.21756774187088013, "rewards/total_composite/mean": 0.7134602665901184, "rewards/total_composite/std": 0.21302269399166107, "reward": 0.7134602665901184, "reward_std": 0.21302267909049988, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1083531305193901, "sampling/sampling_logp_difference/max": 1.336371898651123, "sampling/importance_sampling_ratio/min": 0.26279738545417786, "sampling/importance_sampling_ratio/mean": 1.006940484046936, "sampling/importance_sampling_ratio/max": 1.6972180604934692, "entropy": 0.7061746343970299, "clip_ratio/low_mean": 0.1073880223557353, "clip_ratio/low_min": 0.1073880223557353, "clip_ratio/high_mean": 0.03640110045671463, "clip_ratio/high_max": 0.03640110045671463, "clip_ratio/region_mean": 0.14378912281244993, "reward_total_mean": 0.7134602665901184, "reward_meter_mean": 0.7428567409515381, "reward_meter_std": 0.3768162429332733, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.8025000095367432, "reward_judge_quality_std": 0.21756774187088013, "reward_total_composite_mean": 0.7134602665901184, "reward_total_composite_std": 0.21302269399166107} {"timestamp_utc": "2026-04-13T10:15:57Z", "mode": "train", "global_step": 1177, "epoch": 0.11823204419889503, "loss": -0.1547, "grad_norm": 22.922889709472656, "learning_rate": 6.436363636363637e-06, "num_tokens": 2080972.0, "completions/mean_length": 21.75, "completions/min_length": 14.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.75, "completions/min_terminated_length": 14.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.5803344249725342, "rewards/meter/std": 0.4434331953525543, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9334905743598938, "rewards/repeat_soft/std": 0.06022332236170769, "rewards/judge_quality/mean": 0.41749998927116394, "rewards/judge_quality/std": 0.06902380287647247, "rewards/total_composite/mean": 0.493272066116333, "rewards/total_composite/std": 0.11983183771371841, "reward": 0.493272066116333, "reward_std": 0.11983183771371841, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17677754163742065, "sampling/sampling_logp_difference/max": 1.2432184219360352, "sampling/importance_sampling_ratio/min": 0.2884543538093567, "sampling/importance_sampling_ratio/mean": 1.046566367149353, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0221305936574936, "clip_ratio/low_mean": 0.10992063768208027, "clip_ratio/low_min": 0.10992063768208027, "clip_ratio/high_mean": 0.09235384874045849, "clip_ratio/high_max": 0.09235384874045849, "clip_ratio/region_mean": 0.20227448642253876, "reward_total_mean": 0.493272066116333, "reward_meter_mean": 0.5803344249725342, "reward_meter_std": 0.4434331953525543, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9334905743598938, "reward_repeat_soft_std": 0.06022332236170769, "reward_judge_quality_mean": 0.41749998927116394, "reward_judge_quality_std": 0.06902380287647247, "reward_total_composite_mean": 0.493272066116333, "reward_total_composite_std": 0.11983183771371841} {"timestamp_utc": "2026-04-13T10:16:04Z", "mode": "train", "global_step": 1178, "epoch": 0.11833249623304871, "loss": 0.027, "grad_norm": 11.848631858825684, "learning_rate": 6.433333333333333e-06, "num_tokens": 2082710.0, "completions/mean_length": 48.25, "completions/min_length": 42.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.25, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.7625800967216492, "rewards/meter/std": 0.2791503965854645, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9629368185997009, "rewards/repeat_soft/std": 0.04091007634997368, "rewards/judge_quality/mean": 0.6449999809265137, "rewards/judge_quality/std": 0.24928471446037292, "rewards/total_composite/mean": 0.6047937870025635, "rewards/total_composite/std": 0.1743733137845993, "reward": 0.6047937870025635, "reward_std": 0.1743733137845993, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12717106938362122, "sampling/sampling_logp_difference/max": 2.013184070587158, "sampling/importance_sampling_ratio/min": 0.13356272876262665, "sampling/importance_sampling_ratio/mean": 1.0019422769546509, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.581928551197052, "clip_ratio/low_mean": 0.06915176194161177, "clip_ratio/low_min": 0.06915176194161177, "clip_ratio/high_mean": 0.03293771017342806, "clip_ratio/high_max": 0.03293771017342806, "clip_ratio/region_mean": 0.10208947211503983, "reward_total_mean": 0.6047937870025635, "reward_meter_mean": 0.7625800967216492, "reward_meter_std": 0.2791503965854645, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9629368185997009, "reward_repeat_soft_std": 0.04091007634997368, "reward_judge_quality_mean": 0.6449999809265137, "reward_judge_quality_std": 0.24928471446037292, "reward_total_composite_mean": 0.6047937870025635, "reward_total_composite_std": 0.1743733137845993} {"timestamp_utc": "2026-04-13T10:16:12Z", "mode": "train", "global_step": 1179, "epoch": 0.11843294826720241, "loss": 0.0509, "grad_norm": 13.395605087280273, "learning_rate": 6.430303030303031e-06, "num_tokens": 2084447.0, "completions/mean_length": 49.125, "completions/min_length": 42.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.125, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.8613986968994141, "rewards/meter/std": 0.1988326907157898, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9677962064743042, "rewards/repeat_soft/std": 0.05451434105634689, "rewards/judge_quality/mean": 0.4724999964237213, "rewards/judge_quality/std": 0.11310552060604095, "rewards/total_composite/mean": 0.605843722820282, "rewards/total_composite/std": 0.07325293868780136, "reward": 0.605843722820282, "reward_std": 0.07325293123722076, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1554005891084671, "sampling/sampling_logp_difference/max": 1.7847270965576172, "sampling/importance_sampling_ratio/min": 0.16784286499023438, "sampling/importance_sampling_ratio/mean": 1.0024949312210083, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1742741093039513, "clip_ratio/low_mean": 0.023413379210978746, "clip_ratio/low_min": 0.023413379210978746, "clip_ratio/high_mean": 0.12985734874382615, "clip_ratio/high_max": 0.12985734874382615, "clip_ratio/region_mean": 0.1532707279548049, "reward_total_mean": 0.605843722820282, "reward_meter_mean": 0.8613986968994141, "reward_meter_std": 0.1988326907157898, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9677962064743042, "reward_repeat_soft_std": 0.05451434105634689, "reward_judge_quality_mean": 0.4724999964237213, "reward_judge_quality_std": 0.11310552060604095, "reward_total_composite_mean": 0.605843722820282, "reward_total_composite_std": 0.07325293868780136} {"timestamp_utc": "2026-04-13T10:16:18Z", "mode": "train", "global_step": 1180, "epoch": 0.1185334003013561, "loss": 0.065, "grad_norm": 12.635881423950195, "learning_rate": 6.427272727272728e-06, "num_tokens": 2086088.0, "completions/mean_length": 42.125, "completions/min_length": 31.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.125, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.8736544847488403, "rewards/meter/std": 0.129952535033226, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9728291630744934, "rewards/repeat_soft/std": 0.022369252517819405, "rewards/judge_quality/mean": 0.45499998331069946, "rewards/judge_quality/std": 0.12739142775535583, "rewards/total_composite/mean": 0.6076606512069702, "rewards/total_composite/std": 0.09539171308279037, "reward": 0.6076606512069702, "reward_std": 0.09539171308279037, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17623835802078247, "sampling/sampling_logp_difference/max": 1.3247051239013672, "sampling/importance_sampling_ratio/min": 0.26588135957717896, "sampling/importance_sampling_ratio/mean": 1.0084788799285889, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.268315128982067, "clip_ratio/low_mean": 0.06542681530117989, "clip_ratio/low_min": 0.06542681530117989, "clip_ratio/high_mean": 0.08177647786214948, "clip_ratio/high_max": 0.08177647786214948, "clip_ratio/region_mean": 0.14720329316332936, "reward_total_mean": 0.6076606512069702, "reward_meter_mean": 0.8736544847488403, "reward_meter_std": 0.129952535033226, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9728291630744934, "reward_repeat_soft_std": 0.022369252517819405, "reward_judge_quality_mean": 0.45499998331069946, "reward_judge_quality_std": 0.12739142775535583, "reward_total_composite_mean": 0.6076606512069702, "reward_total_composite_std": 0.09539171308279037} {"timestamp_utc": "2026-04-13T10:16:25Z", "mode": "train", "global_step": 1181, "epoch": 0.1186338523355098, "loss": 0.014, "grad_norm": 10.029298782348633, "learning_rate": 6.424242424242425e-06, "num_tokens": 2088503.0, "completions/mean_length": 79.875, "completions/min_length": 76.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.875, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.9892352819442749, "rewards/meter/std": 0.002484049880877137, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9803063869476318, "rewards/repeat_soft/std": 0.014201012440025806, "rewards/judge_quality/mean": 0.3812499940395355, "rewards/judge_quality/std": 0.08166787773370743, "rewards/total_composite/mean": 0.5421372652053833, "rewards/total_composite/std": 0.05115868151187897, "reward": 0.5421372652053833, "reward_std": 0.05115867778658867, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15114262700080872, "sampling/sampling_logp_difference/max": 1.9406929016113281, "sampling/importance_sampling_ratio/min": 0.14360441267490387, "sampling/importance_sampling_ratio/mean": 1.02043616771698, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8896564543247223, "clip_ratio/low_mean": 0.032595694065093994, "clip_ratio/low_min": 0.032595694065093994, "clip_ratio/high_mean": 0.09571088943630457, "clip_ratio/high_max": 0.09571088943630457, "clip_ratio/region_mean": 0.12830658350139856, "reward_total_mean": 0.5421372652053833, "reward_meter_mean": 0.9892352819442749, "reward_meter_std": 0.002484049880877137, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9803063869476318, "reward_repeat_soft_std": 0.014201012440025806, "reward_judge_quality_mean": 0.3812499940395355, "reward_judge_quality_std": 0.08166787773370743, "reward_total_composite_mean": 0.5421372652053833, "reward_total_composite_std": 0.05115868151187897} {"timestamp_utc": "2026-04-13T10:16:31Z", "mode": "train", "global_step": 1182, "epoch": 0.11873430436966348, "loss": 0.0771, "grad_norm": 60.17795944213867, "learning_rate": 6.4212121212121215e-06, "num_tokens": 2089810.0, "completions/mean_length": 18.375, "completions/min_length": 15.0, "completions/max_length": 25.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 18.375, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 25.0, "rewards/meter/mean": 0.9784665107727051, "rewards/meter/std": 0.019904328510165215, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9497883319854736, "rewards/repeat_soft/std": 0.018914852291345596, "rewards/judge_quality/mean": 0.44624999165534973, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.6262871026992798, "rewards/total_composite/std": 0.01159062422811985, "reward": 0.6262871026992798, "reward_std": 0.011590617708861828, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12480399757623672, "sampling/sampling_logp_difference/max": 1.5895189046859741, "sampling/importance_sampling_ratio/min": 0.20402373373508453, "sampling/importance_sampling_ratio/mean": 1.0191082954406738, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7400579452514648, "clip_ratio/low_mean": 0.04062500037252903, "clip_ratio/low_min": 0.04062500037252903, "clip_ratio/high_mean": 0.037738095968961716, "clip_ratio/high_max": 0.037738095968961716, "clip_ratio/region_mean": 0.07836309634149075, "reward_total_mean": 0.6262871026992798, "reward_meter_mean": 0.9784665107727051, "reward_meter_std": 0.019904328510165215, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9497883319854736, "reward_repeat_soft_std": 0.018914852291345596, "reward_judge_quality_mean": 0.44624999165534973, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.6262871026992798, "reward_total_composite_std": 0.01159062422811985} {"timestamp_utc": "2026-04-13T10:16:38Z", "mode": "train", "global_step": 1183, "epoch": 0.11883475640381717, "loss": 0.0269, "grad_norm": 15.561697006225586, "learning_rate": 6.418181818181819e-06, "num_tokens": 2091487.0, "completions/mean_length": 33.625, "completions/min_length": 30.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.625, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.920923113822937, "rewards/meter/std": 0.044823601841926575, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9764682054519653, "rewards/repeat_soft/std": 0.02194131538271904, "rewards/judge_quality/mean": 0.44749999046325684, "rewards/judge_quality/std": 0.12848013639450073, "rewards/total_composite/mean": 0.6128696799278259, "rewards/total_composite/std": 0.06621865183115005, "reward": 0.6128696799278259, "reward_std": 0.06621863692998886, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12935635447502136, "sampling/sampling_logp_difference/max": 1.3288724422454834, "sampling/importance_sampling_ratio/min": 0.2647756338119507, "sampling/importance_sampling_ratio/mean": 1.0072948932647705, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7739732190966606, "clip_ratio/low_mean": 0.07058823620900512, "clip_ratio/low_min": 0.07058823620900512, "clip_ratio/high_mean": 0.03732493054121733, "clip_ratio/high_max": 0.03732493054121733, "clip_ratio/region_mean": 0.10791316675022244, "reward_total_mean": 0.6128696799278259, "reward_meter_mean": 0.920923113822937, "reward_meter_std": 0.044823601841926575, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9764682054519653, "reward_repeat_soft_std": 0.02194131538271904, "reward_judge_quality_mean": 0.44749999046325684, "reward_judge_quality_std": 0.12848013639450073, "reward_total_composite_mean": 0.6128696799278259, "reward_total_composite_std": 0.06621865183115005} {"timestamp_utc": "2026-04-13T10:16:45Z", "mode": "train", "global_step": 1184, "epoch": 0.11893520843797087, "loss": 0.0227, "grad_norm": 14.760201454162598, "learning_rate": 6.415151515151515e-06, "num_tokens": 2093134.0, "completions/mean_length": 40.875, "completions/min_length": 37.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.875, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.9717105627059937, "rewards/meter/std": 0.03349750488996506, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9868008494377136, "rewards/repeat_soft/std": 0.010895688086748123, "rewards/judge_quality/mean": 0.5900000333786011, "rewards/judge_quality/std": 0.22696760296821594, "rewards/total_composite/mean": 0.7206075191497803, "rewards/total_composite/std": 0.14404785633087158, "reward": 0.7206075191497803, "reward_std": 0.14404785633087158, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15891246497631073, "sampling/sampling_logp_difference/max": 1.3936738967895508, "sampling/importance_sampling_ratio/min": 0.2481618970632553, "sampling/importance_sampling_ratio/mean": 1.0071918964385986, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1980241239070892, "clip_ratio/low_mean": 0.12017614394426346, "clip_ratio/low_min": 0.12017614394426346, "clip_ratio/high_mean": 0.05986295826733112, "clip_ratio/high_max": 0.05986295826733112, "clip_ratio/region_mean": 0.18003910221159458, "reward_total_mean": 0.7206075191497803, "reward_meter_mean": 0.9717105627059937, "reward_meter_std": 0.03349750488996506, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9868008494377136, "reward_repeat_soft_std": 0.010895688086748123, "reward_judge_quality_mean": 0.5900000333786011, "reward_judge_quality_std": 0.22696760296821594, "reward_total_composite_mean": 0.7206075191497803, "reward_total_composite_std": 0.14404785633087158} {"timestamp_utc": "2026-04-13T10:16:51Z", "mode": "train", "global_step": 1185, "epoch": 0.11903566047212456, "loss": -0.0049, "grad_norm": 16.361007690429688, "learning_rate": 6.412121212121213e-06, "num_tokens": 2094701.0, "completions/mean_length": 43.875, "completions/min_length": 38.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.875, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.7773550152778625, "rewards/meter/std": 0.3657516539096832, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9184046983718872, "rewards/repeat_soft/std": 0.06233862787485123, "rewards/judge_quality/mean": 0.5237500071525574, "rewards/judge_quality/std": 0.25150617957115173, "rewards/total_composite/mean": 0.6161110401153564, "rewards/total_composite/std": 0.20731280744075775, "reward": 0.6161110401153564, "reward_std": 0.20731279253959656, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14475561678409576, "sampling/sampling_logp_difference/max": 2.3401975631713867, "sampling/importance_sampling_ratio/min": 0.09630860388278961, "sampling/importance_sampling_ratio/mean": 1.0052721500396729, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0590978786349297, "clip_ratio/low_mean": 0.06890421081334352, "clip_ratio/low_min": 0.06890421081334352, "clip_ratio/high_mean": 0.031746032647788525, "clip_ratio/high_max": 0.031746032647788525, "clip_ratio/region_mean": 0.10065024346113205, "reward_total_mean": 0.6161110401153564, "reward_meter_mean": 0.7773550152778625, "reward_meter_std": 0.3657516539096832, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9184046983718872, "reward_repeat_soft_std": 0.06233862787485123, "reward_judge_quality_mean": 0.5237500071525574, "reward_judge_quality_std": 0.25150617957115173, "reward_total_composite_mean": 0.6161110401153564, "reward_total_composite_std": 0.20731280744075775} {"timestamp_utc": "2026-04-13T10:17:03Z", "mode": "train", "global_step": 1186, "epoch": 0.11913611250627826, "loss": -0.1406, "grad_norm": 2.0124919414520264, "learning_rate": 6.40909090909091e-06, "num_tokens": 2096340.0, "completions/mean_length": 106.875, "completions/min_length": 40.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 49.000003814697266, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9718071222305298, "rewards/meter/std": 0.026004765182733536, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9764590263366699, "rewards/repeat_soft/std": 0.006248929537832737, "rewards/judge_quality/mean": 0.39625000953674316, "rewards/judge_quality/std": 0.14029940962791443, "rewards/total_composite/mean": 0.5499769449234009, "rewards/total_composite/std": 0.22244948148727417, "reward": 0.5499769449234009, "reward_std": 0.22244949638843536, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1452147215604782, "sampling/sampling_logp_difference/max": 1.3545036315917969, "sampling/importance_sampling_ratio/min": 0.2580753564834595, "sampling/importance_sampling_ratio/mean": 1.0315886735916138, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2035367637872696, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.12365831341594458, "clip_ratio/high_max": 0.12365831341594458, "clip_ratio/region_mean": 0.12365831341594458, "reward_total_mean": 0.5499769449234009, "reward_meter_mean": 0.9718071222305298, "reward_meter_std": 0.026004765182733536, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9764590263366699, "reward_repeat_soft_std": 0.006248929537832737, "reward_judge_quality_mean": 0.39625000953674316, "reward_judge_quality_std": 0.14029940962791443, "reward_total_composite_mean": 0.5499769449234009, "reward_total_composite_std": 0.22244948148727417} {"timestamp_utc": "2026-04-13T10:17:15Z", "mode": "train", "global_step": 1187, "epoch": 0.11923656454043194, "loss": -0.1604, "grad_norm": 4.2426886558532715, "learning_rate": 6.406060606060607e-06, "num_tokens": 2098159.0, "completions/mean_length": 124.375, "completions/min_length": 47.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 69.0, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.44092947244644165, "rewards/meter/std": 0.3489566147327423, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9642821550369263, "rewards/repeat_soft/std": 0.04394963011145592, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.22947145998477936, "rewards/total_composite/mean": 0.45709294080734253, "rewards/total_composite/std": 0.2502937316894531, "reward": 0.45709294080734253, "reward_std": 0.25029370188713074, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13997969031333923, "sampling/sampling_logp_difference/max": 1.2993731498718262, "sampling/importance_sampling_ratio/min": 0.272702693939209, "sampling/importance_sampling_ratio/mean": 1.0237239599227905, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8705766573548317, "clip_ratio/low_mean": 0.07386473566293716, "clip_ratio/low_min": 0.07386473566293716, "clip_ratio/high_mean": 0.035914814099669456, "clip_ratio/high_max": 0.035914814099669456, "clip_ratio/region_mean": 0.10977954976260662, "reward_total_mean": 0.45709294080734253, "reward_meter_mean": 0.44092947244644165, "reward_meter_std": 0.3489566147327423, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9642821550369263, "reward_repeat_soft_std": 0.04394963011145592, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.22947145998477936, "reward_total_composite_mean": 0.45709294080734253, "reward_total_composite_std": 0.2502937316894531} {"timestamp_utc": "2026-04-13T10:17:21Z", "mode": "train", "global_step": 1188, "epoch": 0.11933701657458563, "loss": 0.0424, "grad_norm": 12.961089134216309, "learning_rate": 6.403030303030303e-06, "num_tokens": 2100051.0, "completions/mean_length": 55.5, "completions/min_length": 50.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.5, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9832470417022705, "rewards/meter/std": 0.01328179519623518, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9672013521194458, "rewards/repeat_soft/std": 0.020161520689725876, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.22720351815223694, "rewards/total_composite/mean": 0.6988184452056885, "rewards/total_composite/std": 0.1485285609960556, "reward": 0.6988184452056885, "reward_std": 0.1485285609960556, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14231687784194946, "sampling/sampling_logp_difference/max": 1.3428674936294556, "sampling/importance_sampling_ratio/min": 0.26109591126441956, "sampling/importance_sampling_ratio/mean": 1.02568781375885, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.107128344476223, "clip_ratio/low_mean": 0.10461588110774755, "clip_ratio/low_min": 0.10461588110774755, "clip_ratio/high_mean": 0.02727272640913725, "clip_ratio/high_max": 0.02727272640913725, "clip_ratio/region_mean": 0.1318886075168848, "reward_total_mean": 0.6988184452056885, "reward_meter_mean": 0.9832470417022705, "reward_meter_std": 0.01328179519623518, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9672013521194458, "reward_repeat_soft_std": 0.020161520689725876, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.22720351815223694, "reward_total_composite_mean": 0.6988184452056885, "reward_total_composite_std": 0.1485285609960556} {"timestamp_utc": "2026-04-13T10:17:28Z", "mode": "train", "global_step": 1189, "epoch": 0.11943746860873933, "loss": -0.0018, "grad_norm": 10.991680145263672, "learning_rate": 6.4000000000000006e-06, "num_tokens": 2101752.0, "completions/mean_length": 55.625, "completions/min_length": 44.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.625, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9614124298095703, "rewards/meter/std": 0.06390176713466644, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9865628480911255, "rewards/repeat_soft/std": 0.017839282751083374, "rewards/judge_quality/mean": 0.8062500357627869, "rewards/judge_quality/std": 0.16860245168209076, "rewards/total_composite/mean": 0.8503421545028687, "rewards/total_composite/std": 0.10541792213916779, "reward": 0.8503421545028687, "reward_std": 0.10541790723800659, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13659781217575073, "sampling/sampling_logp_difference/max": 1.2765960693359375, "sampling/importance_sampling_ratio/min": 0.27898532152175903, "sampling/importance_sampling_ratio/mean": 1.0041850805282593, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9991523623466492, "clip_ratio/low_mean": 0.08090193755924702, "clip_ratio/low_min": 0.08090193755924702, "clip_ratio/high_mean": 0.06871948949992657, "clip_ratio/high_max": 0.06871948949992657, "clip_ratio/region_mean": 0.14962142705917358, "reward_total_mean": 0.8503421545028687, "reward_meter_mean": 0.9614124298095703, "reward_meter_std": 0.06390176713466644, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9865628480911255, "reward_repeat_soft_std": 0.017839282751083374, "reward_judge_quality_mean": 0.8062500357627869, "reward_judge_quality_std": 0.16860245168209076, "reward_total_composite_mean": 0.8503421545028687, "reward_total_composite_std": 0.10541792213916779} {"timestamp_utc": "2026-04-13T10:17:34Z", "mode": "train", "global_step": 1190, "epoch": 0.11953792064289302, "loss": 0.0464, "grad_norm": 14.893120765686035, "learning_rate": 6.396969696969697e-06, "num_tokens": 2103394.0, "completions/mean_length": 51.25, "completions/min_length": 47.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.25, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.983229398727417, "rewards/meter/std": 0.011052844114601612, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9670378565788269, "rewards/repeat_soft/std": 0.033053942024707794, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.12906256318092346, "rewards/total_composite/mean": 0.6262995004653931, "rewards/total_composite/std": 0.08399740606546402, "reward": 0.6262995004653931, "reward_std": 0.08399741351604462, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15653058886528015, "sampling/sampling_logp_difference/max": 1.4083642959594727, "sampling/importance_sampling_ratio/min": 0.2445429563522339, "sampling/importance_sampling_ratio/mean": 1.0138646364212036, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1422750800848007, "clip_ratio/low_mean": 0.09773599030449986, "clip_ratio/low_min": 0.09773599030449986, "clip_ratio/high_mean": 0.039455581456422806, "clip_ratio/high_max": 0.039455581456422806, "clip_ratio/region_mean": 0.13719157176092267, "reward_total_mean": 0.6262995004653931, "reward_meter_mean": 0.983229398727417, "reward_meter_std": 0.011052844114601612, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9670378565788269, "reward_repeat_soft_std": 0.033053942024707794, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.12906256318092346, "reward_total_composite_mean": 0.6262995004653931, "reward_total_composite_std": 0.08399740606546402} {"timestamp_utc": "2026-04-13T10:17:41Z", "mode": "train", "global_step": 1191, "epoch": 0.11963837267704672, "loss": 0.051, "grad_norm": 10.140250205993652, "learning_rate": 6.393939393939394e-06, "num_tokens": 2105598.0, "completions/mean_length": 88.5, "completions/min_length": 73.0, "completions/max_length": 113.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 88.5, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 113.0, "rewards/meter/mean": 0.9199206829071045, "rewards/meter/std": 0.12989304959774017, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9780960083007812, "rewards/repeat_soft/std": 0.016376817598938942, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.10507651418447495, "rewards/total_composite/mean": 0.5732858180999756, "rewards/total_composite/std": 0.07956836372613907, "reward": 0.5732858180999756, "reward_std": 0.07956835627555847, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14494368433952332, "sampling/sampling_logp_difference/max": 1.548975944519043, "sampling/importance_sampling_ratio/min": 0.21246543526649475, "sampling/importance_sampling_ratio/mean": 1.0270488262176514, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1746059358119965, "clip_ratio/low_mean": 0.12986954487860203, "clip_ratio/low_min": 0.12986954487860203, "clip_ratio/high_mean": 0.0150602413341403, "clip_ratio/high_max": 0.0150602413341403, "clip_ratio/region_mean": 0.14492978621274233, "reward_total_mean": 0.5732858180999756, "reward_meter_mean": 0.9199206829071045, "reward_meter_std": 0.12989304959774017, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9780960083007812, "reward_repeat_soft_std": 0.016376817598938942, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.10507651418447495, "reward_total_composite_mean": 0.5732858180999756, "reward_total_composite_std": 0.07956836372613907} {"timestamp_utc": "2026-04-13T10:17:48Z", "mode": "train", "global_step": 1192, "epoch": 0.1197388247112004, "loss": 0.0775, "grad_norm": 12.860295295715332, "learning_rate": 6.390909090909091e-06, "num_tokens": 2107245.0, "completions/mean_length": 51.875, "completions/min_length": 39.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.875, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.7144586443901062, "rewards/meter/std": 0.37646904587745667, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9928861260414124, "rewards/repeat_soft/std": 0.014112521894276142, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.5860370993614197, "rewards/total_composite/std": 0.17091825604438782, "reward": 0.5860370993614197, "reward_std": 0.17091824114322662, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14491435885429382, "sampling/sampling_logp_difference/max": 1.269066333770752, "sampling/importance_sampling_ratio/min": 0.281093955039978, "sampling/importance_sampling_ratio/mean": 1.0024237632751465, "sampling/importance_sampling_ratio/max": 1.8214935064315796, "entropy": 1.239496372640133, "clip_ratio/low_mean": 0.05775345675647259, "clip_ratio/low_min": 0.05775345675647259, "clip_ratio/high_mean": 0.0726571804843843, "clip_ratio/high_max": 0.0726571804843843, "clip_ratio/region_mean": 0.13041063724085689, "reward_total_mean": 0.5860370993614197, "reward_meter_mean": 0.7144586443901062, "reward_meter_std": 0.37646904587745667, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9928861260414124, "reward_repeat_soft_std": 0.014112521894276142, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.5860370993614197, "reward_total_composite_std": 0.17091825604438782} {"timestamp_utc": "2026-04-13T10:17:54Z", "mode": "train", "global_step": 1193, "epoch": 0.11983927674535409, "loss": 0.1202, "grad_norm": 13.604833602905273, "learning_rate": 6.387878787878789e-06, "num_tokens": 2108993.0, "completions/mean_length": 44.5, "completions/min_length": 26.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.5, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.6406293511390686, "rewards/meter/std": 0.36187151074409485, "rewards/count_adherence/mean": 0.8333333730697632, "rewards/count_adherence/std": 0.17817415297031403, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9930717945098877, "rewards/repeat_soft/std": 0.006495763082057238, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.0843462198972702, "rewards/total_composite/mean": 0.4684458076953888, "rewards/total_composite/std": 0.07827360928058624, "reward": 0.4684458076953888, "reward_std": 0.07827360183000565, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17882013320922852, "sampling/sampling_logp_difference/max": 2.479620933532715, "sampling/importance_sampling_ratio/min": 0.08377497643232346, "sampling/importance_sampling_ratio/mean": 0.98419189453125, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.017261803150177, "clip_ratio/low_mean": 0.049816531129181385, "clip_ratio/low_min": 0.049816531129181385, "clip_ratio/high_mean": 0.1255369170103222, "clip_ratio/high_max": 0.1255369170103222, "clip_ratio/region_mean": 0.1753534481395036, "reward_total_mean": 0.4684458076953888, "reward_meter_mean": 0.6406293511390686, "reward_meter_std": 0.36187151074409485, "reward_count_adherence_mean": 0.8333333730697632, "reward_count_adherence_std": 0.17817415297031403, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9930717945098877, "reward_repeat_soft_std": 0.006495763082057238, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.0843462198972702, "reward_total_composite_mean": 0.4684458076953888, "reward_total_composite_std": 0.07827360928058624} {"timestamp_utc": "2026-04-13T10:18:00Z", "mode": "train", "global_step": 1194, "epoch": 0.11993972877950779, "loss": 0.0689, "grad_norm": 19.21489143371582, "learning_rate": 6.384848484848485e-06, "num_tokens": 2110251.0, "completions/mean_length": 21.25, "completions/min_length": 17.0, "completions/max_length": 27.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.25, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.91082763671875, "rewards/meter/std": 0.15774142742156982, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9571585655212402, "rewards/repeat_soft/std": 0.01005425676703453, "rewards/judge_quality/mean": 0.59375, "rewards/judge_quality/std": 0.2775370180606842, "rewards/total_composite/mean": 0.6985938549041748, "rewards/total_composite/std": 0.1734265834093094, "reward": 0.6985938549041748, "reward_std": 0.1734265685081482, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15081371366977692, "sampling/sampling_logp_difference/max": 1.801047921180725, "sampling/importance_sampling_ratio/min": 0.165125772356987, "sampling/importance_sampling_ratio/mean": 1.0244626998901367, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1096824705600739, "clip_ratio/low_mean": 0.07454019039869308, "clip_ratio/low_min": 0.07454019039869308, "clip_ratio/high_mean": 0.07072027772665024, "clip_ratio/high_max": 0.07072027772665024, "clip_ratio/region_mean": 0.14526046812534332, "reward_total_mean": 0.6985938549041748, "reward_meter_mean": 0.91082763671875, "reward_meter_std": 0.15774142742156982, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9571585655212402, "reward_repeat_soft_std": 0.01005425676703453, "reward_judge_quality_mean": 0.59375, "reward_judge_quality_std": 0.2775370180606842, "reward_total_composite_mean": 0.6985938549041748, "reward_total_composite_std": 0.1734265834093094} {"timestamp_utc": "2026-04-13T10:18:06Z", "mode": "train", "global_step": 1195, "epoch": 0.12004018081366148, "loss": 0.0729, "grad_norm": 21.43175506591797, "learning_rate": 6.381818181818182e-06, "num_tokens": 2111668.0, "completions/mean_length": 27.125, "completions/min_length": 22.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.125, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.7170892357826233, "rewards/meter/std": 0.4193773567676544, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9456349611282349, "rewards/repeat_soft/std": 0.02970712073147297, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.5472277402877808, "rewards/total_composite/std": 0.1188732162117958, "reward": 0.5472277402877808, "reward_std": 0.1188732162117958, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19483132660388947, "sampling/sampling_logp_difference/max": 1.9253969192504883, "sampling/importance_sampling_ratio/min": 0.1458178609609604, "sampling/importance_sampling_ratio/mean": 1.0215253829956055, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1635321155190468, "clip_ratio/low_mean": 0.03362068999558687, "clip_ratio/low_min": 0.03362068999558687, "clip_ratio/high_mean": 0.13018791005015373, "clip_ratio/high_max": 0.13018791005015373, "clip_ratio/region_mean": 0.1638086000457406, "reward_total_mean": 0.5472277402877808, "reward_meter_mean": 0.7170892357826233, "reward_meter_std": 0.4193773567676544, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9456349611282349, "reward_repeat_soft_std": 0.02970712073147297, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.5472277402877808, "reward_total_composite_std": 0.1188732162117958} {"timestamp_utc": "2026-04-13T10:18:12Z", "mode": "train", "global_step": 1196, "epoch": 0.12014063284781516, "loss": 0.0023, "grad_norm": 22.521121978759766, "learning_rate": 6.37878787878788e-06, "num_tokens": 2113405.0, "completions/mean_length": 41.125, "completions/min_length": 34.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.125, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.8774996995925903, "rewards/meter/std": 0.20249055325984955, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9252956509590149, "rewards/repeat_soft/std": 0.041638702154159546, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.5887066721916199, "rewards/total_composite/std": 0.054475292563438416, "reward": 0.5887066721916199, "reward_std": 0.054475292563438416, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1631079614162445, "sampling/sampling_logp_difference/max": 1.6264400482177734, "sampling/importance_sampling_ratio/min": 0.19662831723690033, "sampling/importance_sampling_ratio/mean": 1.0326478481292725, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0275902897119522, "clip_ratio/low_mean": 0.04046817682683468, "clip_ratio/low_min": 0.04046817682683468, "clip_ratio/high_mean": 0.10343743953853846, "clip_ratio/high_max": 0.10343743953853846, "clip_ratio/region_mean": 0.14390561636537313, "reward_total_mean": 0.5887066721916199, "reward_meter_mean": 0.8774996995925903, "reward_meter_std": 0.20249055325984955, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9252956509590149, "reward_repeat_soft_std": 0.041638702154159546, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.5887066721916199, "reward_total_composite_std": 0.054475292563438416} {"timestamp_utc": "2026-04-13T10:18:19Z", "mode": "train", "global_step": 1197, "epoch": 0.12024108488196886, "loss": 0.108, "grad_norm": 14.894396781921387, "learning_rate": 6.375757575757576e-06, "num_tokens": 2115159.0, "completions/mean_length": 44.25, "completions/min_length": 34.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.25, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.8855868577957153, "rewards/meter/std": 0.16638562083244324, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9769302010536194, "rewards/repeat_soft/std": 0.028060367330908775, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.5948513746261597, "rewards/total_composite/std": 0.049289148300886154, "reward": 0.5948513746261597, "reward_std": 0.04928916320204735, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1516123265028, "sampling/sampling_logp_difference/max": 2.5276780128479004, "sampling/importance_sampling_ratio/min": 0.07984420657157898, "sampling/importance_sampling_ratio/mean": 1.0203876495361328, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8630895391106606, "clip_ratio/low_mean": 0.030555556528270245, "clip_ratio/low_min": 0.030555556528270245, "clip_ratio/high_mean": 0.07832845905795693, "clip_ratio/high_max": 0.07832845905795693, "clip_ratio/region_mean": 0.10888401558622718, "reward_total_mean": 0.5948513746261597, "reward_meter_mean": 0.8855868577957153, "reward_meter_std": 0.16638562083244324, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9769302010536194, "reward_repeat_soft_std": 0.028060367330908775, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.5948513746261597, "reward_total_composite_std": 0.049289148300886154} {"timestamp_utc": "2026-04-13T10:18:25Z", "mode": "train", "global_step": 1198, "epoch": 0.12034153691612255, "loss": -0.0627, "grad_norm": 12.346944808959961, "learning_rate": 6.372727272727274e-06, "num_tokens": 2117026.0, "completions/mean_length": 59.375, "completions/min_length": 50.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.375, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.8909122943878174, "rewards/meter/std": 0.14564402401447296, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9201802611351013, "rewards/repeat_soft/std": 0.05797836557030678, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.5553241968154907, "rewards/total_composite/std": 0.08962288498878479, "reward": 0.5553241968154907, "reward_std": 0.089622862637043, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15406766533851624, "sampling/sampling_logp_difference/max": 2.4812746047973633, "sampling/importance_sampling_ratio/min": 0.08363655209541321, "sampling/importance_sampling_ratio/mean": 1.0228735208511353, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7829513102769852, "clip_ratio/low_mean": 0.11219498421996832, "clip_ratio/low_min": 0.11219498421996832, "clip_ratio/high_mean": 0.04119047708809376, "clip_ratio/high_max": 0.04119047708809376, "clip_ratio/region_mean": 0.15338546130806208, "reward_total_mean": 0.5553241968154907, "reward_meter_mean": 0.8909122943878174, "reward_meter_std": 0.14564402401447296, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9201802611351013, "reward_repeat_soft_std": 0.05797836557030678, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.5553241968154907, "reward_total_composite_std": 0.08962288498878479} {"timestamp_utc": "2026-04-13T10:18:31Z", "mode": "train", "global_step": 1199, "epoch": 0.12044198895027625, "loss": -0.0039, "grad_norm": 12.607721328735352, "learning_rate": 6.3696969696969706e-06, "num_tokens": 2118740.0, "completions/mean_length": 45.25, "completions/min_length": 36.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.25, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9817730784416199, "rewards/meter/std": 0.014380814507603645, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8610401153564453, "rewards/repeat_soft/std": 0.04926241561770439, "rewards/judge_quality/mean": 0.6737500429153442, "rewards/judge_quality/std": 0.263435423374176, "rewards/total_composite/mean": 0.7600692510604858, "rewards/total_composite/std": 0.17443729937076569, "reward": 0.7600692510604858, "reward_std": 0.1744372844696045, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12888361513614655, "sampling/sampling_logp_difference/max": 1.837851643562317, "sampling/importance_sampling_ratio/min": 0.15915898978710175, "sampling/importance_sampling_ratio/mean": 1.0331822633743286, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9980748891830444, "clip_ratio/low_mean": 0.0693011898547411, "clip_ratio/low_min": 0.0693011898547411, "clip_ratio/high_mean": 0.05967627977952361, "clip_ratio/high_max": 0.05967627977952361, "clip_ratio/region_mean": 0.1289774696342647, "reward_total_mean": 0.7600692510604858, "reward_meter_mean": 0.9817730784416199, "reward_meter_std": 0.014380814507603645, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8610401153564453, "reward_repeat_soft_std": 0.04926241561770439, "reward_judge_quality_mean": 0.6737500429153442, "reward_judge_quality_std": 0.263435423374176, "reward_total_composite_mean": 0.7600692510604858, "reward_total_composite_std": 0.17443729937076569} {"timestamp_utc": "2026-04-13T10:18:37Z", "mode": "train", "global_step": 1200, "epoch": 0.12054244098442994, "loss": 0.0559, "grad_norm": 12.950010299682617, "learning_rate": 6.366666666666668e-06, "num_tokens": 2120393.0, "completions/mean_length": 46.625, "completions/min_length": 41.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.625, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.8502044677734375, "rewards/meter/std": 0.1859995424747467, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9613869190216064, "rewards/repeat_soft/std": 0.028637342154979706, "rewards/judge_quality/mean": 0.48624998331069946, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.6189001798629761, "rewards/total_composite/std": 0.14047762751579285, "reward": 0.6189001798629761, "reward_std": 0.14047759771347046, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1475149244070053, "sampling/sampling_logp_difference/max": 1.9337434768676758, "sampling/importance_sampling_ratio/min": 0.1446058601140976, "sampling/importance_sampling_ratio/mean": 1.035650610923767, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.14615136384964, "clip_ratio/low_mean": 0.08140896027907729, "clip_ratio/low_min": 0.08140896027907729, "clip_ratio/high_mean": 0.0376881156116724, "clip_ratio/high_max": 0.0376881156116724, "clip_ratio/region_mean": 0.11909707589074969, "reward_total_mean": 0.6189001798629761, "reward_meter_mean": 0.8502044677734375, "reward_meter_std": 0.1859995424747467, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9613869190216064, "reward_repeat_soft_std": 0.028637342154979706, "reward_judge_quality_mean": 0.48624998331069946, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.6189001798629761, "reward_total_composite_std": 0.14047762751579285} {"timestamp_utc": "2026-04-13T10:19:46Z", "mode": "eval", "global_step": 1200, "epoch": 0.12054244098442994, "eval_loss": NaN, "eval_runtime": 68.5981, "eval_samples_per_second": 1.166, "eval_steps_per_second": 0.146, "eval_num_tokens": 2120393.0, "eval_completions/mean_length": 98.275, "eval_completions/min_length": 30.7, "eval_completions/max_length": 349.6, "eval_completions/clipped_ratio": 0.0875, "eval_completions/mean_terminated_length": 58.62321586608887, "eval_completions/min_terminated_length": 30.7, "eval_completions/max_terminated_length": 92.1, "eval_rewards/meter/mean": 0.6437475919723511, "eval_rewards/meter/std": 0.378156253695488, "eval_rewards/count_adherence/mean": 0.8241666734218598, "eval_rewards/count_adherence/std": 0.15415611043572425, "eval_rewards/hard_gate/mean": 0.9125, "eval_rewards/hard_gate/std": 0.2230676978826523, "eval_rewards/repeat_soft/mean": 0.9744892477989197, "eval_rewards/repeat_soft/std": 0.030129980575293303, "eval_rewards/judge_quality/mean": 0.45325000286102296, "eval_rewards/judge_quality/std": 0.19670253470540047, "eval_rewards/total_composite/mean": 0.4832571804523468, "eval_rewards/total_composite/std": 0.19506984874606131, "eval_reward": 0.4832571804523468, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.08069026917219162, "eval_sampling/sampling_logp_difference/max": 0.9795767307281494, "eval_sampling/importance_sampling_ratio/min": 0.38638010025024416, "eval_sampling/importance_sampling_ratio/mean": 1.0223345279693603, "eval_sampling/importance_sampling_ratio/max": 1.4525879859924316, "eval_entropy": 0.9961097061634063, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.4832571804523468, "eval_reward_meter_mean": 0.6437475919723511, "eval_reward_meter_std": 0.378156253695488, "eval_reward_count_adherence_mean": 0.8241666734218598, "eval_reward_count_adherence_std": 0.15415611043572425, "eval_reward_hard_gate_mean": 0.9125, "eval_reward_hard_gate_std": 0.2230676978826523, "eval_reward_repeat_soft_mean": 0.9744892477989197, "eval_reward_repeat_soft_std": 0.030129980575293303, "eval_reward_judge_quality_mean": 0.45325000286102296, "eval_reward_judge_quality_std": 0.19670253470540047, "eval_reward_total_composite_mean": 0.4832571804523468, "eval_reward_total_composite_std": 0.19506984874606131} {"timestamp_utc": "2026-04-13T10:20:01Z", "mode": "train", "global_step": 1201, "epoch": 0.12064289301858362, "loss": -0.1373, "grad_norm": 3.051819086074829, "learning_rate": 6.363636363636364e-06, "num_tokens": 2122638.0, "completions/mean_length": 134.625, "completions/min_length": 58.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 80.71428680419922, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.7561625242233276, "rewards/meter/std": 0.3270168900489807, "rewards/count_adherence/mean": 0.6500000357627869, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9848114252090454, "rewards/repeat_soft/std": 0.00961005873978138, "rewards/judge_quality/mean": 0.3824999928474426, "rewards/judge_quality/std": 0.1060660108923912, "rewards/total_composite/mean": 0.4379623830318451, "rewards/total_composite/std": 0.19056583940982819, "reward": 0.4379623830318451, "reward_std": 0.190565824508667, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15711110830307007, "sampling/sampling_logp_difference/max": 1.6434390544891357, "sampling/importance_sampling_ratio/min": 0.1933140754699707, "sampling/importance_sampling_ratio/mean": 1.002091407775879, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8832895681262016, "clip_ratio/low_mean": 0.024122806265950203, "clip_ratio/low_min": 0.024122806265950203, "clip_ratio/high_mean": 0.114609788171947, "clip_ratio/high_max": 0.114609788171947, "clip_ratio/region_mean": 0.1387325944378972, "reward_total_mean": 0.4379623830318451, "reward_meter_mean": 0.7561625242233276, "reward_meter_std": 0.3270168900489807, "reward_count_adherence_mean": 0.6500000357627869, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9848114252090454, "reward_repeat_soft_std": 0.00961005873978138, "reward_judge_quality_mean": 0.3824999928474426, "reward_judge_quality_std": 0.1060660108923912, "reward_total_composite_mean": 0.4379623830318451, "reward_total_composite_std": 0.19056583940982819} {"timestamp_utc": "2026-04-13T10:20:07Z", "mode": "train", "global_step": 1202, "epoch": 0.12074334505273732, "loss": 0.1207, "grad_norm": 33.19580841064453, "learning_rate": 6.3606060606060615e-06, "num_tokens": 2124021.0, "completions/mean_length": 19.875, "completions/min_length": 15.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 19.875, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.8972835540771484, "rewards/meter/std": 0.1867736577987671, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9558625221252441, "rewards/repeat_soft/std": 0.018056858330965042, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.5894105434417725, "rewards/total_composite/std": 0.047464992851018906, "reward": 0.5894105434417725, "reward_std": 0.047464992851018906, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16092872619628906, "sampling/sampling_logp_difference/max": 1.2862911224365234, "sampling/importance_sampling_ratio/min": 0.27629363536834717, "sampling/importance_sampling_ratio/mean": 1.0243053436279297, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0373342111706734, "clip_ratio/low_mean": 0.02604166604578495, "clip_ratio/low_min": 0.02604166604578495, "clip_ratio/high_mean": 0.12085863808169961, "clip_ratio/high_max": 0.12085863808169961, "clip_ratio/region_mean": 0.14690030412748456, "reward_total_mean": 0.5894105434417725, "reward_meter_mean": 0.8972835540771484, "reward_meter_std": 0.1867736577987671, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9558625221252441, "reward_repeat_soft_std": 0.018056858330965042, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.5894105434417725, "reward_total_composite_std": 0.047464992851018906} {"timestamp_utc": "2026-04-13T10:20:13Z", "mode": "train", "global_step": 1203, "epoch": 0.12084379708689101, "loss": -0.0348, "grad_norm": 27.95676040649414, "learning_rate": 6.357575757575758e-06, "num_tokens": 2125408.0, "completions/mean_length": 19.375, "completions/min_length": 14.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 19.375, "completions/min_terminated_length": 14.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.6801613569259644, "rewards/meter/std": 0.3801872134208679, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9488004446029663, "rewards/repeat_soft/std": 0.03874802216887474, "rewards/judge_quality/mean": 0.36374998092651367, "rewards/judge_quality/std": 0.09500939399003983, "rewards/total_composite/mean": 0.5100680589675903, "rewards/total_composite/std": 0.1137004867196083, "reward": 0.5100680589675903, "reward_std": 0.11370047926902771, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2133399397134781, "sampling/sampling_logp_difference/max": 1.2526464462280273, "sampling/importance_sampling_ratio/min": 0.28574758768081665, "sampling/importance_sampling_ratio/mean": 1.0028443336486816, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0662638172507286, "clip_ratio/low_mean": 0.0587418326176703, "clip_ratio/low_min": 0.0587418326176703, "clip_ratio/high_mean": 0.09089940413832664, "clip_ratio/high_max": 0.09089940413832664, "clip_ratio/region_mean": 0.14964123675599694, "reward_total_mean": 0.5100680589675903, "reward_meter_mean": 0.6801613569259644, "reward_meter_std": 0.3801872134208679, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9488004446029663, "reward_repeat_soft_std": 0.03874802216887474, "reward_judge_quality_mean": 0.36374998092651367, "reward_judge_quality_std": 0.09500939399003983, "reward_total_composite_mean": 0.5100680589675903, "reward_total_composite_std": 0.1137004867196083} {"timestamp_utc": "2026-04-13T10:20:24Z", "mode": "train", "global_step": 1204, "epoch": 0.1209442491210447, "loss": -0.1478, "grad_norm": 2.8766109943389893, "learning_rate": 6.354545454545455e-06, "num_tokens": 2127018.0, "completions/mean_length": 120.25, "completions/min_length": 54.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 64.28572082519531, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.8735039234161377, "rewards/meter/std": 0.2158215194940567, "rewards/count_adherence/mean": 0.6875, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9839870929718018, "rewards/repeat_soft/std": 0.011796806007623672, "rewards/judge_quality/mean": 0.3100000023841858, "rewards/judge_quality/std": 0.1345893144607544, "rewards/total_composite/mean": 0.4356952905654907, "rewards/total_composite/std": 0.19193559885025024, "reward": 0.4356952905654907, "reward_std": 0.19193559885025024, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15613046288490295, "sampling/sampling_logp_difference/max": 1.7565383911132812, "sampling/importance_sampling_ratio/min": 0.17264144122600555, "sampling/importance_sampling_ratio/mean": 1.0262750387191772, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2565559148788452, "clip_ratio/low_mean": 0.02321428619325161, "clip_ratio/low_min": 0.02321428619325161, "clip_ratio/high_mean": 0.12299090530723333, "clip_ratio/high_max": 0.12299090530723333, "clip_ratio/region_mean": 0.14620519150048494, "reward_total_mean": 0.4356952905654907, "reward_meter_mean": 0.8735039234161377, "reward_meter_std": 0.2158215194940567, "reward_count_adherence_mean": 0.6875, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9839870929718018, "reward_repeat_soft_std": 0.011796806007623672, "reward_judge_quality_mean": 0.3100000023841858, "reward_judge_quality_std": 0.1345893144607544, "reward_total_composite_mean": 0.4356952905654907, "reward_total_composite_std": 0.19193559885025024} {"timestamp_utc": "2026-04-13T10:20:35Z", "mode": "train", "global_step": 1205, "epoch": 0.12104470115519839, "loss": 0.0408, "grad_norm": 16.52765655517578, "learning_rate": 6.3515151515151516e-06, "num_tokens": 2128647.0, "completions/mean_length": 43.625, "completions/min_length": 41.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.625, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.6900322437286377, "rewards/meter/std": 0.3865191638469696, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9763786792755127, "rewards/repeat_soft/std": 0.022137310355901718, "rewards/judge_quality/mean": 0.7150000333786011, "rewards/judge_quality/std": 0.2887411117553711, "rewards/total_composite/mean": 0.649908721446991, "rewards/total_composite/std": 0.2352641224861145, "reward": 0.649908721446991, "reward_std": 0.2352641224861145, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.151315838098526, "sampling/sampling_logp_difference/max": 1.98641037940979, "sampling/importance_sampling_ratio/min": 0.13718698918819427, "sampling/importance_sampling_ratio/mean": 1.007492184638977, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5421228930354118, "clip_ratio/low_mean": 0.06591028068214655, "clip_ratio/low_min": 0.06591028068214655, "clip_ratio/high_mean": 0.04702952038496733, "clip_ratio/high_max": 0.04702952038496733, "clip_ratio/region_mean": 0.11293980106711388, "reward_total_mean": 0.649908721446991, "reward_meter_mean": 0.6900322437286377, "reward_meter_std": 0.3865191638469696, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9763786792755127, "reward_repeat_soft_std": 0.022137310355901718, "reward_judge_quality_mean": 0.7150000333786011, "reward_judge_quality_std": 0.2887411117553711, "reward_total_composite_mean": 0.649908721446991, "reward_total_composite_std": 0.2352641224861145} {"timestamp_utc": "2026-04-13T10:20:42Z", "mode": "train", "global_step": 1206, "epoch": 0.12114515318935208, "loss": 0.0696, "grad_norm": 9.657299995422363, "learning_rate": 6.34848484848485e-06, "num_tokens": 2130488.0, "completions/mean_length": 52.125, "completions/min_length": 42.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.125, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9881328344345093, "rewards/meter/std": 0.006366685964167118, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8367646932601929, "rewards/repeat_soft/std": 0.07589803636074066, "rewards/judge_quality/mean": 0.46875, "rewards/judge_quality/std": 0.19334924221038818, "rewards/total_composite/mean": 0.6268348097801208, "rewards/total_composite/std": 0.1335621327161789, "reward": 0.6268348097801208, "reward_std": 0.1335621327161789, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12683773040771484, "sampling/sampling_logp_difference/max": 1.4757649898529053, "sampling/importance_sampling_ratio/min": 0.33859530091285706, "sampling/importance_sampling_ratio/mean": 1.0261094570159912, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0749919340014458, "clip_ratio/low_mean": 0.09005030803382397, "clip_ratio/low_min": 0.09005030803382397, "clip_ratio/high_mean": 0.02380952425301075, "clip_ratio/high_max": 0.02380952425301075, "clip_ratio/region_mean": 0.11385983228683472, "reward_total_mean": 0.6268348097801208, "reward_meter_mean": 0.9881328344345093, "reward_meter_std": 0.006366685964167118, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8367646932601929, "reward_repeat_soft_std": 0.07589803636074066, "reward_judge_quality_mean": 0.46875, "reward_judge_quality_std": 0.19334924221038818, "reward_total_composite_mean": 0.6268348097801208, "reward_total_composite_std": 0.1335621327161789} {"timestamp_utc": "2026-04-13T10:20:48Z", "mode": "train", "global_step": 1207, "epoch": 0.12124560522350578, "loss": 0.0543, "grad_norm": 29.118677139282227, "learning_rate": 6.345454545454546e-06, "num_tokens": 2132026.0, "completions/mean_length": 25.25, "completions/min_length": 22.0, "completions/max_length": 27.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.25, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.8704559803009033, "rewards/meter/std": 0.26886969804763794, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.948982834815979, "rewards/repeat_soft/std": 0.03707736358046532, "rewards/judge_quality/mean": 0.3762499988079071, "rewards/judge_quality/std": 0.11287634819746017, "rewards/total_composite/mean": 0.5548781156539917, "rewards/total_composite/std": 0.1002330482006073, "reward": 0.5548781156539917, "reward_std": 0.1002330407500267, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1961924135684967, "sampling/sampling_logp_difference/max": 1.9085369110107422, "sampling/importance_sampling_ratio/min": 0.14829720556735992, "sampling/importance_sampling_ratio/mean": 1.0134321451187134, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2682722434401512, "clip_ratio/low_mean": 0.029230769723653793, "clip_ratio/low_min": 0.029230769723653793, "clip_ratio/high_mean": 0.11837542243301868, "clip_ratio/high_max": 0.11837542243301868, "clip_ratio/region_mean": 0.14760619215667248, "reward_total_mean": 0.5548781156539917, "reward_meter_mean": 0.8704559803009033, "reward_meter_std": 0.26886969804763794, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.948982834815979, "reward_repeat_soft_std": 0.03707736358046532, "reward_judge_quality_mean": 0.3762499988079071, "reward_judge_quality_std": 0.11287634819746017, "reward_total_composite_mean": 0.5548781156539917, "reward_total_composite_std": 0.1002330482006073} {"timestamp_utc": "2026-04-13T10:21:00Z", "mode": "train", "global_step": 1208, "epoch": 0.12134605725765947, "loss": -0.1173, "grad_norm": 2.0062859058380127, "learning_rate": 6.342424242424243e-06, "num_tokens": 2133776.0, "completions/mean_length": 167.75, "completions/min_length": 33.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 53.0, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.740172803401947, "rewards/meter/std": 0.41915446519851685, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.17251639068126678, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9885010719299316, "rewards/repeat_soft/std": 0.01072147861123085, "rewards/judge_quality/mean": 0.5712499618530273, "rewards/judge_quality/std": 0.4018328785896301, "rewards/total_composite/mean": 0.6020362973213196, "rewards/total_composite/std": 0.40308788418769836, "reward": 0.6020362973213196, "reward_std": 0.40308788418769836, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13050059974193573, "sampling/sampling_logp_difference/max": 1.5308656692504883, "sampling/importance_sampling_ratio/min": 0.2163483202457428, "sampling/importance_sampling_ratio/mean": 1.0152531862258911, "sampling/importance_sampling_ratio/max": 1.7760437726974487, "entropy": 0.87591902166605, "clip_ratio/low_mean": 0.028764205053448677, "clip_ratio/low_min": 0.028764205053448677, "clip_ratio/high_mean": 0.047134652733802795, "clip_ratio/high_max": 0.047134652733802795, "clip_ratio/region_mean": 0.07589885778725147, "reward_total_mean": 0.6020362973213196, "reward_meter_mean": 0.740172803401947, "reward_meter_std": 0.41915446519851685, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.17251639068126678, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9885010719299316, "reward_repeat_soft_std": 0.01072147861123085, "reward_judge_quality_mean": 0.5712499618530273, "reward_judge_quality_std": 0.4018328785896301, "reward_total_composite_mean": 0.6020362973213196, "reward_total_composite_std": 0.40308788418769836} {"timestamp_utc": "2026-04-13T10:21:12Z", "mode": "train", "global_step": 1209, "epoch": 0.12144650929181317, "loss": -0.124, "grad_norm": 3.3813726902008057, "learning_rate": 6.33939393939394e-06, "num_tokens": 2135597.0, "completions/mean_length": 115.625, "completions/min_length": 49.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 59.000003814697266, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.6542345285415649, "rewards/meter/std": 0.41005656123161316, "rewards/count_adherence/mean": 0.7000000476837158, "rewards/count_adherence/std": 0.10690449178218842, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9741815328598022, "rewards/repeat_soft/std": 0.022501269355416298, "rewards/judge_quality/mean": 0.3499999940395355, "rewards/judge_quality/std": 0.15445756912231445, "rewards/total_composite/mean": 0.41418832540512085, "rewards/total_composite/std": 0.19170942902565002, "reward": 0.41418832540512085, "reward_std": 0.19170941412448883, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1486818790435791, "sampling/sampling_logp_difference/max": 1.7528820037841797, "sampling/importance_sampling_ratio/min": 0.17327384650707245, "sampling/importance_sampling_ratio/mean": 1.0278178453445435, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.713676430284977, "clip_ratio/low_mean": 0.05007684417068958, "clip_ratio/low_min": 0.05007684417068958, "clip_ratio/high_mean": 0.08644945221021771, "clip_ratio/high_max": 0.08644945221021771, "clip_ratio/region_mean": 0.1365262963809073, "reward_total_mean": 0.41418832540512085, "reward_meter_mean": 0.6542345285415649, "reward_meter_std": 0.41005656123161316, "reward_count_adherence_mean": 0.7000000476837158, "reward_count_adherence_std": 0.10690449178218842, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9741815328598022, "reward_repeat_soft_std": 0.022501269355416298, "reward_judge_quality_mean": 0.3499999940395355, "reward_judge_quality_std": 0.15445756912231445, "reward_total_composite_mean": 0.41418832540512085, "reward_total_composite_std": 0.19170942902565002} {"timestamp_utc": "2026-04-13T10:21:18Z", "mode": "train", "global_step": 1210, "epoch": 0.12154696132596685, "loss": 0.1422, "grad_norm": 18.450532913208008, "learning_rate": 6.336363636363637e-06, "num_tokens": 2136956.0, "completions/mean_length": 23.875, "completions/min_length": 14.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.875, "completions/min_terminated_length": 14.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.7502397298812866, "rewards/meter/std": 0.3737407624721527, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9594065546989441, "rewards/repeat_soft/std": 0.008749538101255894, "rewards/judge_quality/mean": 0.7987500429153442, "rewards/judge_quality/std": 0.22465452551841736, "rewards/total_composite/mean": 0.7429962158203125, "rewards/total_composite/std": 0.237124502658844, "reward": 0.7429962158203125, "reward_std": 0.2371244877576828, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1377735584974289, "sampling/sampling_logp_difference/max": 1.2020378112792969, "sampling/importance_sampling_ratio/min": 0.30058106780052185, "sampling/importance_sampling_ratio/mean": 1.015731692314148, "sampling/importance_sampling_ratio/max": 1.8095163106918335, "entropy": 0.9119813740253448, "clip_ratio/low_mean": 0.030828773509711027, "clip_ratio/low_min": 0.030828773509711027, "clip_ratio/high_mean": 0.08687951043248177, "clip_ratio/high_max": 0.08687951043248177, "clip_ratio/region_mean": 0.11770828394219279, "reward_total_mean": 0.7429962158203125, "reward_meter_mean": 0.7502397298812866, "reward_meter_std": 0.3737407624721527, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9594065546989441, "reward_repeat_soft_std": 0.008749538101255894, "reward_judge_quality_mean": 0.7987500429153442, "reward_judge_quality_std": 0.22465452551841736, "reward_total_composite_mean": 0.7429962158203125, "reward_total_composite_std": 0.237124502658844} {"timestamp_utc": "2026-04-13T10:21:29Z", "mode": "train", "global_step": 1211, "epoch": 0.12164741336012054, "loss": -0.0765, "grad_norm": 2.047220468521118, "learning_rate": 6.333333333333333e-06, "num_tokens": 2138377.0, "completions/mean_length": 152.625, "completions/min_length": 28.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 32.833335876464844, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.3923335075378418, "rewards/meter/std": 0.39527276158332825, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9937474727630615, "rewards/repeat_soft/std": 0.011514980345964432, "rewards/judge_quality/mean": 0.45249998569488525, "rewards/judge_quality/std": 0.33065953850746155, "rewards/total_composite/mean": 0.39616629481315613, "rewards/total_composite/std": 0.2613082528114319, "reward": 0.39616629481315613, "reward_std": 0.2613082528114319, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14189504086971283, "sampling/sampling_logp_difference/max": 1.5270648002624512, "sampling/importance_sampling_ratio/min": 0.21717219054698944, "sampling/importance_sampling_ratio/mean": 1.041178584098816, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7455131411552429, "clip_ratio/low_mean": 0.010135134682059288, "clip_ratio/low_min": 0.010135134682059288, "clip_ratio/high_mean": 0.06774892006069422, "clip_ratio/high_max": 0.06774892006069422, "clip_ratio/region_mean": 0.0778840547427535, "reward_total_mean": 0.39616629481315613, "reward_meter_mean": 0.3923335075378418, "reward_meter_std": 0.39527276158332825, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9937474727630615, "reward_repeat_soft_std": 0.011514980345964432, "reward_judge_quality_mean": 0.45249998569488525, "reward_judge_quality_std": 0.33065953850746155, "reward_total_composite_mean": 0.39616629481315613, "reward_total_composite_std": 0.2613082528114319} {"timestamp_utc": "2026-04-13T10:21:36Z", "mode": "train", "global_step": 1212, "epoch": 0.12174786539427424, "loss": 0.0196, "grad_norm": 9.696742057800293, "learning_rate": 6.330303030303031e-06, "num_tokens": 2140550.0, "completions/mean_length": 77.625, "completions/min_length": 70.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.625, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.9477736353874207, "rewards/meter/std": 0.08169277012348175, "rewards/count_adherence/mean": 0.6000000238418579, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9340115785598755, "rewards/repeat_soft/std": 0.04480970278382301, "rewards/judge_quality/mean": 0.4937500059604645, "rewards/judge_quality/std": 0.1728696972131729, "rewards/total_composite/mean": 0.5658751726150513, "rewards/total_composite/std": 0.12052886933088303, "reward": 0.5658751726150513, "reward_std": 0.12052884697914124, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14696049690246582, "sampling/sampling_logp_difference/max": 1.8601083755493164, "sampling/importance_sampling_ratio/min": 0.15565575659275055, "sampling/importance_sampling_ratio/mean": 1.0185673236846924, "sampling/importance_sampling_ratio/max": 1.997831106185913, "entropy": 1.1255382373929024, "clip_ratio/low_mean": 0.11282453499734402, "clip_ratio/low_min": 0.11282453499734402, "clip_ratio/high_mean": 0.017123287543654442, "clip_ratio/high_max": 0.017123287543654442, "clip_ratio/region_mean": 0.12994782254099846, "reward_total_mean": 0.5658751726150513, "reward_meter_mean": 0.9477736353874207, "reward_meter_std": 0.08169277012348175, "reward_count_adherence_mean": 0.6000000238418579, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9340115785598755, "reward_repeat_soft_std": 0.04480970278382301, "reward_judge_quality_mean": 0.4937500059604645, "reward_judge_quality_std": 0.1728696972131729, "reward_total_composite_mean": 0.5658751726150513, "reward_total_composite_std": 0.12052886933088303} {"timestamp_utc": "2026-04-13T10:21:42Z", "mode": "train", "global_step": 1213, "epoch": 0.12184831742842793, "loss": 0.0752, "grad_norm": 24.94639778137207, "learning_rate": 6.327272727272727e-06, "num_tokens": 2142167.0, "completions/mean_length": 36.125, "completions/min_length": 31.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.125, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.8696717023849487, "rewards/meter/std": 0.14692988991737366, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9133477210998535, "rewards/repeat_soft/std": 0.03822728246450424, "rewards/judge_quality/mean": 0.4137499928474426, "rewards/judge_quality/std": 0.06781013309955597, "rewards/total_composite/mean": 0.5706425905227661, "rewards/total_composite/std": 0.05210505425930023, "reward": 0.5706425905227661, "reward_std": 0.05210505798459053, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12412320077419281, "sampling/sampling_logp_difference/max": 1.1363520622253418, "sampling/importance_sampling_ratio/min": 0.3209878206253052, "sampling/importance_sampling_ratio/mean": 1.0149773359298706, "sampling/importance_sampling_ratio/max": 1.9755158424377441, "entropy": 0.82471963763237, "clip_ratio/low_mean": 0.06491832248866558, "clip_ratio/low_min": 0.06491832248866558, "clip_ratio/high_mean": 0.07793703209608793, "clip_ratio/high_max": 0.07793703209608793, "clip_ratio/region_mean": 0.1428553545847535, "reward_total_mean": 0.5706425905227661, "reward_meter_mean": 0.8696717023849487, "reward_meter_std": 0.14692988991737366, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9133477210998535, "reward_repeat_soft_std": 0.03822728246450424, "reward_judge_quality_mean": 0.4137499928474426, "reward_judge_quality_std": 0.06781013309955597, "reward_total_composite_mean": 0.5706425905227661, "reward_total_composite_std": 0.05210505425930023} {"timestamp_utc": "2026-04-13T10:21:53Z", "mode": "train", "global_step": 1214, "epoch": 0.12194876946258161, "loss": -0.0485, "grad_norm": 6.261666774749756, "learning_rate": 6.324242424242425e-06, "num_tokens": 2143780.0, "completions/mean_length": 100.625, "completions/min_length": 38.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 41.85714340209961, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.23466385900974274, "rewards/meter/std": 0.17924705147743225, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9908305406570435, "rewards/repeat_soft/std": 0.009137889370322227, "rewards/judge_quality/mean": 0.45499998331069946, "rewards/judge_quality/std": 0.2334829568862915, "rewards/total_composite/mean": 0.41426029801368713, "rewards/total_composite/std": 0.054228782653808594, "reward": 0.41426029801368713, "reward_std": 0.05422878637909889, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1751175969839096, "sampling/sampling_logp_difference/max": 1.8739943504333496, "sampling/importance_sampling_ratio/min": 0.153509259223938, "sampling/importance_sampling_ratio/mean": 0.9763612151145935, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7590757980942726, "clip_ratio/low_mean": 0.029545455239713192, "clip_ratio/low_min": 0.029545455239713192, "clip_ratio/high_mean": 0.1381261870265007, "clip_ratio/high_max": 0.1381261870265007, "clip_ratio/region_mean": 0.1676716422662139, "reward_total_mean": 0.41426029801368713, "reward_meter_mean": 0.23466385900974274, "reward_meter_std": 0.17924705147743225, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9908305406570435, "reward_repeat_soft_std": 0.009137889370322227, "reward_judge_quality_mean": 0.45499998331069946, "reward_judge_quality_std": 0.2334829568862915, "reward_total_composite_mean": 0.41426029801368713, "reward_total_composite_std": 0.054228782653808594} {"timestamp_utc": "2026-04-13T10:22:00Z", "mode": "train", "global_step": 1215, "epoch": 0.1220492214967353, "loss": 0.0459, "grad_norm": 12.655740737915039, "learning_rate": 6.3212121212121216e-06, "num_tokens": 2145739.0, "completions/mean_length": 63.875, "completions/min_length": 53.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.875, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.6042683124542236, "rewards/meter/std": 0.23978233337402344, "rewards/count_adherence/mean": 0.6000000238418579, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9636364579200745, "rewards/repeat_soft/std": 0.01974622532725334, "rewards/judge_quality/mean": 0.6737500429153442, "rewards/judge_quality/std": 0.10541586577892303, "rewards/total_composite/mean": 0.5332971811294556, "rewards/total_composite/std": 0.1182984784245491, "reward": 0.5332971811294556, "reward_std": 0.1182984858751297, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1559729278087616, "sampling/sampling_logp_difference/max": 1.490548849105835, "sampling/importance_sampling_ratio/min": 0.22524899244308472, "sampling/importance_sampling_ratio/mean": 1.002070665359497, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1542490348219872, "clip_ratio/low_mean": 0.05461149103939533, "clip_ratio/low_min": 0.05461149103939533, "clip_ratio/high_mean": 0.08680233173072338, "clip_ratio/high_max": 0.08680233173072338, "clip_ratio/region_mean": 0.1414138227701187, "reward_total_mean": 0.5332971811294556, "reward_meter_mean": 0.6042683124542236, "reward_meter_std": 0.23978233337402344, "reward_count_adherence_mean": 0.6000000238418579, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9636364579200745, "reward_repeat_soft_std": 0.01974622532725334, "reward_judge_quality_mean": 0.6737500429153442, "reward_judge_quality_std": 0.10541586577892303, "reward_total_composite_mean": 0.5332971811294556, "reward_total_composite_std": 0.1182984784245491} {"timestamp_utc": "2026-04-13T10:22:07Z", "mode": "train", "global_step": 1216, "epoch": 0.122149673530889, "loss": -0.0055, "grad_norm": 16.260692596435547, "learning_rate": 6.318181818181819e-06, "num_tokens": 2147291.0, "completions/mean_length": 42.0, "completions/min_length": 30.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.0, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.985662579536438, "rewards/meter/std": 0.008431307040154934, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9779547452926636, "rewards/repeat_soft/std": 0.015956100076436996, "rewards/judge_quality/mean": 0.5062500238418579, "rewards/judge_quality/std": 0.21084441244602203, "rewards/total_composite/mean": 0.6705451011657715, "rewards/total_composite/std": 0.1335761845111847, "reward": 0.6705451011657715, "reward_std": 0.1335761994123459, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13670815527439117, "sampling/sampling_logp_difference/max": 1.4380803108215332, "sampling/importance_sampling_ratio/min": 0.23738302290439606, "sampling/importance_sampling_ratio/mean": 1.0070143938064575, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9411209970712662, "clip_ratio/low_mean": 0.07499094074591994, "clip_ratio/low_min": 0.07499094074591994, "clip_ratio/high_mean": 0.01648550760000944, "clip_ratio/high_max": 0.01648550760000944, "clip_ratio/region_mean": 0.09147644834592938, "reward_total_mean": 0.6705451011657715, "reward_meter_mean": 0.985662579536438, "reward_meter_std": 0.008431307040154934, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9779547452926636, "reward_repeat_soft_std": 0.015956100076436996, "reward_judge_quality_mean": 0.5062500238418579, "reward_judge_quality_std": 0.21084441244602203, "reward_total_composite_mean": 0.6705451011657715, "reward_total_composite_std": 0.1335761845111847} {"timestamp_utc": "2026-04-13T10:22:13Z", "mode": "train", "global_step": 1217, "epoch": 0.1222501255650427, "loss": 0.0067, "grad_norm": 13.174397468566895, "learning_rate": 6.315151515151515e-06, "num_tokens": 2148780.0, "completions/mean_length": 44.125, "completions/min_length": 32.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.125, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.568111777305603, "rewards/meter/std": 0.25833097100257874, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.968593955039978, "rewards/repeat_soft/std": 0.024218551814556122, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.48826876282691956, "rewards/total_composite/std": 0.09525391459465027, "reward": 0.48826876282691956, "reward_std": 0.09525391459465027, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17072643339633942, "sampling/sampling_logp_difference/max": 1.8180158138275146, "sampling/importance_sampling_ratio/min": 0.16234755516052246, "sampling/importance_sampling_ratio/mean": 1.0049854516983032, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.038679413497448, "clip_ratio/low_mean": 0.06209849356673658, "clip_ratio/low_min": 0.06209849356673658, "clip_ratio/high_mean": 0.08722630236297846, "clip_ratio/high_max": 0.08722630236297846, "clip_ratio/region_mean": 0.14932479592971504, "reward_total_mean": 0.48826876282691956, "reward_meter_mean": 0.568111777305603, "reward_meter_std": 0.25833097100257874, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.968593955039978, "reward_repeat_soft_std": 0.024218551814556122, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.48826876282691956, "reward_total_composite_std": 0.09525391459465027} {"timestamp_utc": "2026-04-13T10:22:19Z", "mode": "train", "global_step": 1218, "epoch": 0.12235057759919639, "loss": 0.098, "grad_norm": 11.035683631896973, "learning_rate": 6.3121212121212125e-06, "num_tokens": 2150629.0, "completions/mean_length": 73.125, "completions/min_length": 59.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.125, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.7245287895202637, "rewards/meter/std": 0.37711793184280396, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9799288511276245, "rewards/repeat_soft/std": 0.01665422134101391, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.49478572607040405, "rewards/total_composite/std": 0.10128509998321533, "reward": 0.49478572607040405, "reward_std": 0.10128511488437653, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13519148528575897, "sampling/sampling_logp_difference/max": 2.9129977226257324, "sampling/importance_sampling_ratio/min": 0.05431266874074936, "sampling/importance_sampling_ratio/mean": 1.0262887477874756, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8518487140536308, "clip_ratio/low_mean": 0.04737659823149443, "clip_ratio/low_min": 0.04737659823149443, "clip_ratio/high_mean": 0.06279222015291452, "clip_ratio/high_max": 0.06279222015291452, "clip_ratio/region_mean": 0.11016881838440895, "reward_total_mean": 0.49478572607040405, "reward_meter_mean": 0.7245287895202637, "reward_meter_std": 0.37711793184280396, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9799288511276245, "reward_repeat_soft_std": 0.01665422134101391, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.49478572607040405, "reward_total_composite_std": 0.10128509998321533} {"timestamp_utc": "2026-04-13T10:22:26Z", "mode": "train", "global_step": 1219, "epoch": 0.12245102963335007, "loss": -0.1254, "grad_norm": 14.57003116607666, "learning_rate": 6.309090909090909e-06, "num_tokens": 2152274.0, "completions/mean_length": 45.625, "completions/min_length": 34.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.625, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.03117642179131508, "rewards/meter/std": 0.05061213672161102, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9878250360488892, "rewards/repeat_soft/std": 0.009674848057329655, "rewards/judge_quality/mean": 0.38999998569488525, "rewards/judge_quality/std": 0.0975411981344223, "rewards/total_composite/mean": 0.318015456199646, "rewards/total_composite/std": 0.022586451843380928, "reward": 0.318015456199646, "reward_std": 0.022586463019251823, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16530752182006836, "sampling/sampling_logp_difference/max": 1.4637900590896606, "sampling/importance_sampling_ratio/min": 0.23135775327682495, "sampling/importance_sampling_ratio/mean": 1.0113688707351685, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9115501157939434, "clip_ratio/low_mean": 0.08679158054292202, "clip_ratio/low_min": 0.08679158054292202, "clip_ratio/high_mean": 0.060084665194153786, "clip_ratio/high_max": 0.060084665194153786, "clip_ratio/region_mean": 0.1468762457370758, "reward_total_mean": 0.318015456199646, "reward_meter_mean": 0.03117642179131508, "reward_meter_std": 0.05061213672161102, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9878250360488892, "reward_repeat_soft_std": 0.009674848057329655, "reward_judge_quality_mean": 0.38999998569488525, "reward_judge_quality_std": 0.0975411981344223, "reward_total_composite_mean": 0.318015456199646, "reward_total_composite_std": 0.022586451843380928} {"timestamp_utc": "2026-04-13T10:22:33Z", "mode": "train", "global_step": 1220, "epoch": 0.12255148166750376, "loss": -0.0312, "grad_norm": 11.163514137268066, "learning_rate": 6.306060606060607e-06, "num_tokens": 2154227.0, "completions/mean_length": 71.125, "completions/min_length": 51.0, "completions/max_length": 95.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.125, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.9727790355682373, "rewards/meter/std": 0.024574285373091698, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9593303203582764, "rewards/repeat_soft/std": 0.05558227747678757, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.13695022463798523, "rewards/total_composite/mean": 0.6099128127098083, "rewards/total_composite/std": 0.09227675199508667, "reward": 0.6099128127098083, "reward_std": 0.09227675199508667, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14367783069610596, "sampling/sampling_logp_difference/max": 1.4354729652404785, "sampling/importance_sampling_ratio/min": 0.23800277709960938, "sampling/importance_sampling_ratio/mean": 1.013994812965393, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9198137000203133, "clip_ratio/low_mean": 0.10493800975382328, "clip_ratio/low_min": 0.10493800975382328, "clip_ratio/high_mean": 0.0327856270596385, "clip_ratio/high_max": 0.0327856270596385, "clip_ratio/region_mean": 0.13772363681346178, "reward_total_mean": 0.6099128127098083, "reward_meter_mean": 0.9727790355682373, "reward_meter_std": 0.024574285373091698, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9593303203582764, "reward_repeat_soft_std": 0.05558227747678757, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.13695022463798523, "reward_total_composite_mean": 0.6099128127098083, "reward_total_composite_std": 0.09227675199508667} {"timestamp_utc": "2026-04-13T10:22:39Z", "mode": "train", "global_step": 1221, "epoch": 0.12265193370165746, "loss": 0.0271, "grad_norm": 12.287282943725586, "learning_rate": 6.303030303030303e-06, "num_tokens": 2155872.0, "completions/mean_length": 51.625, "completions/min_length": 46.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.625, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.5839359164237976, "rewards/meter/std": 0.3622641861438751, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9617362022399902, "rewards/repeat_soft/std": 0.05584303289651871, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.5050953030586243, "rewards/total_composite/std": 0.10229077190160751, "reward": 0.5050953030586243, "reward_std": 0.10229074954986572, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15814530849456787, "sampling/sampling_logp_difference/max": 1.9131996631622314, "sampling/importance_sampling_ratio/min": 0.14760734140872955, "sampling/importance_sampling_ratio/mean": 1.007646918296814, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0089810118079185, "clip_ratio/low_mean": 0.06683951243758202, "clip_ratio/low_min": 0.06683951243758202, "clip_ratio/high_mean": 0.0798441655933857, "clip_ratio/high_max": 0.0798441655933857, "clip_ratio/region_mean": 0.1466836780309677, "reward_total_mean": 0.5050953030586243, "reward_meter_mean": 0.5839359164237976, "reward_meter_std": 0.3622641861438751, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9617362022399902, "reward_repeat_soft_std": 0.05584303289651871, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.5050953030586243, "reward_total_composite_std": 0.10229077190160751} {"timestamp_utc": "2026-04-13T10:22:46Z", "mode": "train", "global_step": 1222, "epoch": 0.12275238573581115, "loss": 0.0436, "grad_norm": 15.896768569946289, "learning_rate": 6.300000000000001e-06, "num_tokens": 2157397.0, "completions/mean_length": 40.625, "completions/min_length": 30.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.625, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.6093494296073914, "rewards/meter/std": 0.3976246118545532, "rewards/count_adherence/mean": 0.7083333730697632, "rewards/count_adherence/std": 0.11785111576318741, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9590468406677246, "rewards/repeat_soft/std": 0.048096902668476105, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.49619966745376587, "rewards/total_composite/std": 0.17820341885089874, "reward": 0.49619966745376587, "reward_std": 0.17820340394973755, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15029707551002502, "sampling/sampling_logp_difference/max": 1.7684967517852783, "sampling/importance_sampling_ratio/min": 0.17058923840522766, "sampling/importance_sampling_ratio/mean": 1.0053178071975708, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9463603720068932, "clip_ratio/low_mean": 0.07149485219269991, "clip_ratio/low_min": 0.07149485219269991, "clip_ratio/high_mean": 0.06642336864024401, "clip_ratio/high_max": 0.06642336864024401, "clip_ratio/region_mean": 0.13791822083294392, "reward_total_mean": 0.49619966745376587, "reward_meter_mean": 0.6093494296073914, "reward_meter_std": 0.3976246118545532, "reward_count_adherence_mean": 0.7083333730697632, "reward_count_adherence_std": 0.11785111576318741, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9590468406677246, "reward_repeat_soft_std": 0.048096902668476105, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.49619966745376587, "reward_total_composite_std": 0.17820341885089874} {"timestamp_utc": "2026-04-13T10:22:52Z", "mode": "train", "global_step": 1223, "epoch": 0.12285283776996485, "loss": 0.0731, "grad_norm": 11.490691184997559, "learning_rate": 6.296969696969697e-06, "num_tokens": 2159255.0, "completions/mean_length": 58.25, "completions/min_length": 47.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.25, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.7078536748886108, "rewards/meter/std": 0.2628180682659149, "rewards/count_adherence/mean": 0.65625, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9808167219161987, "rewards/repeat_soft/std": 0.016820814460515976, "rewards/judge_quality/mean": 0.5349999666213989, "rewards/judge_quality/std": 0.1911618709564209, "rewards/total_composite/mean": 0.531480073928833, "rewards/total_composite/std": 0.13504460453987122, "reward": 0.531480073928833, "reward_std": 0.1350446194410324, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13397450745105743, "sampling/sampling_logp_difference/max": 2.532672643661499, "sampling/importance_sampling_ratio/min": 0.07944640517234802, "sampling/importance_sampling_ratio/mean": 1.0112334489822388, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8477783501148224, "clip_ratio/low_mean": 0.0765253221616149, "clip_ratio/low_min": 0.0765253221616149, "clip_ratio/high_mean": 0.050978997722268105, "clip_ratio/high_max": 0.050978997722268105, "clip_ratio/region_mean": 0.127504319883883, "reward_total_mean": 0.531480073928833, "reward_meter_mean": 0.7078536748886108, "reward_meter_std": 0.2628180682659149, "reward_count_adherence_mean": 0.65625, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9808167219161987, "reward_repeat_soft_std": 0.016820814460515976, "reward_judge_quality_mean": 0.5349999666213989, "reward_judge_quality_std": 0.1911618709564209, "reward_total_composite_mean": 0.531480073928833, "reward_total_composite_std": 0.13504460453987122} {"timestamp_utc": "2026-04-13T10:22:59Z", "mode": "train", "global_step": 1224, "epoch": 0.12295328980411853, "loss": 0.0102, "grad_norm": 11.508916854858398, "learning_rate": 6.293939393939394e-06, "num_tokens": 2161308.0, "completions/mean_length": 69.625, "completions/min_length": 55.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.625, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.9514361023902893, "rewards/meter/std": 0.07141122221946716, "rewards/count_adherence/mean": 0.6000000238418579, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9387154579162598, "rewards/repeat_soft/std": 0.04537630081176758, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5205493569374084, "rewards/total_composite/std": 0.020482094958424568, "reward": 0.5205493569374084, "reward_std": 0.020482098683714867, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1273832470178604, "sampling/sampling_logp_difference/max": 1.6701726913452148, "sampling/importance_sampling_ratio/min": 0.18821455538272858, "sampling/importance_sampling_ratio/mean": 1.0133082866668701, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8839424699544907, "clip_ratio/low_mean": 0.03731842152774334, "clip_ratio/low_min": 0.03731842152774334, "clip_ratio/high_mean": 0.07454585377126932, "clip_ratio/high_max": 0.07454585377126932, "clip_ratio/region_mean": 0.11186427529901266, "reward_total_mean": 0.5205493569374084, "reward_meter_mean": 0.9514361023902893, "reward_meter_std": 0.07141122221946716, "reward_count_adherence_mean": 0.6000000238418579, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9387154579162598, "reward_repeat_soft_std": 0.04537630081176758, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5205493569374084, "reward_total_composite_std": 0.020482094958424568} {"timestamp_utc": "2026-04-13T10:23:05Z", "mode": "train", "global_step": 1225, "epoch": 0.12305374183827222, "loss": 0.0963, "grad_norm": 13.007549285888672, "learning_rate": 6.290909090909092e-06, "num_tokens": 2162785.0, "completions/mean_length": 34.625, "completions/min_length": 30.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.625, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.7389113306999207, "rewards/meter/std": 0.3665609359741211, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.933692216873169, "rewards/repeat_soft/std": 0.06272569298744202, "rewards/judge_quality/mean": 0.5137500166893005, "rewards/judge_quality/std": 0.20777307450771332, "rewards/total_composite/mean": 0.6081284284591675, "rewards/total_composite/std": 0.18631310760974884, "reward": 0.6081284284591675, "reward_std": 0.18631310760974884, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13696837425231934, "sampling/sampling_logp_difference/max": 1.710554599761963, "sampling/importance_sampling_ratio/min": 0.18076550960540771, "sampling/importance_sampling_ratio/mean": 1.015040397644043, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7119800746440887, "clip_ratio/low_mean": 0.07738095428794622, "clip_ratio/low_min": 0.07738095428794622, "clip_ratio/high_mean": 0.04546847240999341, "clip_ratio/high_max": 0.04546847240999341, "clip_ratio/region_mean": 0.12284942669793963, "reward_total_mean": 0.6081284284591675, "reward_meter_mean": 0.7389113306999207, "reward_meter_std": 0.3665609359741211, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.933692216873169, "reward_repeat_soft_std": 0.06272569298744202, "reward_judge_quality_mean": 0.5137500166893005, "reward_judge_quality_std": 0.20777307450771332, "reward_total_composite_mean": 0.6081284284591675, "reward_total_composite_std": 0.18631310760974884} {"timestamp_utc": "2026-04-13T10:23:11Z", "mode": "train", "global_step": 1226, "epoch": 0.12315419387242592, "loss": 0.0561, "grad_norm": 11.328813552856445, "learning_rate": 6.287878787878788e-06, "num_tokens": 2164366.0, "completions/mean_length": 38.625, "completions/min_length": 34.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.625, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.9663524627685547, "rewards/meter/std": 0.048049844801425934, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9799480438232422, "rewards/repeat_soft/std": 0.018092192709445953, "rewards/judge_quality/mean": 0.4937499761581421, "rewards/judge_quality/std": 0.1728696972131729, "rewards/total_composite/mean": 0.657935619354248, "rewards/total_composite/std": 0.11399483680725098, "reward": 0.657935619354248, "reward_std": 0.11399482190608978, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13801391422748566, "sampling/sampling_logp_difference/max": 1.7289395332336426, "sampling/importance_sampling_ratio/min": 0.17747251689434052, "sampling/importance_sampling_ratio/mean": 0.9950167536735535, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.901582732796669, "clip_ratio/low_mean": 0.10738228587433696, "clip_ratio/low_min": 0.10738228587433696, "clip_ratio/high_mean": 0.02302631549537182, "clip_ratio/high_max": 0.02302631549537182, "clip_ratio/region_mean": 0.13040860136970878, "reward_total_mean": 0.657935619354248, "reward_meter_mean": 0.9663524627685547, "reward_meter_std": 0.048049844801425934, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9799480438232422, "reward_repeat_soft_std": 0.018092192709445953, "reward_judge_quality_mean": 0.4937499761581421, "reward_judge_quality_std": 0.1728696972131729, "reward_total_composite_mean": 0.657935619354248, "reward_total_composite_std": 0.11399483680725098} {"timestamp_utc": "2026-04-13T10:23:22Z", "mode": "train", "global_step": 1227, "epoch": 0.12325464590657961, "loss": -0.0594, "grad_norm": 3.061837911605835, "learning_rate": 6.284848484848486e-06, "num_tokens": 2165732.0, "completions/mean_length": 83.75, "completions/min_length": 17.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 22.571430206298828, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.4164186120033264, "rewards/meter/std": 0.4045465290546417, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9296796321868896, "rewards/repeat_soft/std": 0.0432172529399395, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.13617216050624847, "rewards/total_composite/mean": 0.40845227241516113, "rewards/total_composite/std": 0.1992538571357727, "reward": 0.40845227241516113, "reward_std": 0.1992538571357727, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15326885879039764, "sampling/sampling_logp_difference/max": 1.05519437789917, "sampling/importance_sampling_ratio/min": 0.34812474250793457, "sampling/importance_sampling_ratio/mean": 1.0181257724761963, "sampling/importance_sampling_ratio/max": 1.7947677373886108, "entropy": 0.9504754319787025, "clip_ratio/low_mean": 0.034852758049964905, "clip_ratio/low_min": 0.034852758049964905, "clip_ratio/high_mean": 0.047062008175998926, "clip_ratio/high_max": 0.047062008175998926, "clip_ratio/region_mean": 0.08191476622596383, "reward_total_mean": 0.40845227241516113, "reward_meter_mean": 0.4164186120033264, "reward_meter_std": 0.4045465290546417, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9296796321868896, "reward_repeat_soft_std": 0.0432172529399395, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.13617216050624847, "reward_total_composite_mean": 0.40845227241516113, "reward_total_composite_std": 0.1992538571357727} {"timestamp_utc": "2026-04-13T10:23:28Z", "mode": "train", "global_step": 1228, "epoch": 0.1233550979407333, "loss": 0.0354, "grad_norm": 13.09066390991211, "learning_rate": 6.2818181818181825e-06, "num_tokens": 2167365.0, "completions/mean_length": 48.125, "completions/min_length": 40.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.125, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.6213930249214172, "rewards/meter/std": 0.372549831867218, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9939090013504028, "rewards/repeat_soft/std": 0.003991083241999149, "rewards/judge_quality/mean": 0.6525000333786011, "rewards/judge_quality/std": 0.2418234497308731, "rewards/total_composite/mean": 0.5973623394966125, "rewards/total_composite/std": 0.18629467487335205, "reward": 0.5973623394966125, "reward_std": 0.18629464507102966, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17509528994560242, "sampling/sampling_logp_difference/max": 1.982809066772461, "sampling/importance_sampling_ratio/min": 0.13768193125724792, "sampling/importance_sampling_ratio/mean": 1.0242944955825806, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2939858064055443, "clip_ratio/low_mean": 0.08759469725191593, "clip_ratio/low_min": 0.08759469725191593, "clip_ratio/high_mean": 0.0906465258449316, "clip_ratio/high_max": 0.0906465258449316, "clip_ratio/region_mean": 0.17824122309684753, "reward_total_mean": 0.5973623394966125, "reward_meter_mean": 0.6213930249214172, "reward_meter_std": 0.372549831867218, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9939090013504028, "reward_repeat_soft_std": 0.003991083241999149, "reward_judge_quality_mean": 0.6525000333786011, "reward_judge_quality_std": 0.2418234497308731, "reward_total_composite_mean": 0.5973623394966125, "reward_total_composite_std": 0.18629467487335205} {"timestamp_utc": "2026-04-13T10:23:34Z", "mode": "train", "global_step": 1229, "epoch": 0.12345554997488699, "loss": 0.1696, "grad_norm": 27.195316314697266, "learning_rate": 6.27878787878788e-06, "num_tokens": 2168751.0, "completions/mean_length": 26.25, "completions/min_length": 19.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.25, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.47008028626441956, "rewards/meter/std": 0.3700377345085144, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9698743224143982, "rewards/repeat_soft/std": 0.030791107565164566, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.22720351815223694, "rewards/total_composite/mean": 0.5009973049163818, "rewards/total_composite/std": 0.11817308515310287, "reward": 0.5009973049163818, "reward_std": 0.11817307770252228, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1846369206905365, "sampling/sampling_logp_difference/max": 1.955707311630249, "sampling/importance_sampling_ratio/min": 0.14146436750888824, "sampling/importance_sampling_ratio/mean": 0.993920087814331, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9901880025863647, "clip_ratio/low_mean": 0.08829365111887455, "clip_ratio/low_min": 0.08829365111887455, "clip_ratio/high_mean": 0.08385943993926048, "clip_ratio/high_max": 0.08385943993926048, "clip_ratio/region_mean": 0.17215309105813503, "reward_total_mean": 0.5009973049163818, "reward_meter_mean": 0.47008028626441956, "reward_meter_std": 0.3700377345085144, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9698743224143982, "reward_repeat_soft_std": 0.030791107565164566, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.22720351815223694, "reward_total_composite_mean": 0.5009973049163818, "reward_total_composite_std": 0.11817308515310287} {"timestamp_utc": "2026-04-13T10:23:46Z", "mode": "train", "global_step": 1230, "epoch": 0.12355600200904068, "loss": -0.1841, "grad_norm": 2.2863810062408447, "learning_rate": 6.275757575757576e-06, "num_tokens": 2171009.0, "completions/mean_length": 147.25, "completions/min_length": 75.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 95.14286041259766, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 122.0, "rewards/meter/mean": 0.8865006566047668, "rewards/meter/std": 0.1508045494556427, "rewards/count_adherence/mean": 0.5208333730697632, "rewards/count_adherence/std": 0.0589255727827549, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9836091995239258, "rewards/repeat_soft/std": 0.013653808273375034, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.13593590259552002, "rewards/total_composite/mean": 0.4220849275588989, "rewards/total_composite/std": 0.17547102272510529, "reward": 0.4220849275588989, "reward_std": 0.17547102272510529, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1561456173658371, "sampling/sampling_logp_difference/max": 2.5082197189331055, "sampling/importance_sampling_ratio/min": 0.08141304552555084, "sampling/importance_sampling_ratio/mean": 1.0233533382415771, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.137905977666378, "clip_ratio/low_mean": 0.017326733097434044, "clip_ratio/low_min": 0.017326733097434044, "clip_ratio/high_mean": 0.11770072672516108, "clip_ratio/high_max": 0.11770072672516108, "clip_ratio/region_mean": 0.13502745982259512, "reward_total_mean": 0.4220849275588989, "reward_meter_mean": 0.8865006566047668, "reward_meter_std": 0.1508045494556427, "reward_count_adherence_mean": 0.5208333730697632, "reward_count_adherence_std": 0.0589255727827549, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9836091995239258, "reward_repeat_soft_std": 0.013653808273375034, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.13593590259552002, "reward_total_composite_mean": 0.4220849275588989, "reward_total_composite_std": 0.17547102272510529} {"timestamp_utc": "2026-04-13T10:23:53Z", "mode": "train", "global_step": 1231, "epoch": 0.12365645404319438, "loss": 0.06, "grad_norm": 9.06653881072998, "learning_rate": 6.2727272727272734e-06, "num_tokens": 2173136.0, "completions/mean_length": 83.875, "completions/min_length": 60.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 83.875, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.949677586555481, "rewards/meter/std": 0.0947742909193039, "rewards/count_adherence/mean": 0.5416666865348816, "rewards/count_adherence/std": 0.07715168595314026, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9752757549285889, "rewards/repeat_soft/std": 0.015269976109266281, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.5621194839477539, "rewards/total_composite/std": 0.09587524831295013, "reward": 0.5621194839477539, "reward_std": 0.09587525576353073, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14602883160114288, "sampling/sampling_logp_difference/max": 2.093897819519043, "sampling/importance_sampling_ratio/min": 0.12320596724748611, "sampling/importance_sampling_ratio/mean": 1.019765853881836, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1949023455381393, "clip_ratio/low_mean": 0.09645771980285645, "clip_ratio/low_min": 0.09645771980285645, "clip_ratio/high_mean": 0.03787878900766373, "clip_ratio/high_max": 0.03787878900766373, "clip_ratio/region_mean": 0.13433650881052017, "reward_total_mean": 0.5621194839477539, "reward_meter_mean": 0.949677586555481, "reward_meter_std": 0.0947742909193039, "reward_count_adherence_mean": 0.5416666865348816, "reward_count_adherence_std": 0.07715168595314026, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9752757549285889, "reward_repeat_soft_std": 0.015269976109266281, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.5621194839477539, "reward_total_composite_std": 0.09587524831295013} {"timestamp_utc": "2026-04-13T10:24:04Z", "mode": "train", "global_step": 1232, "epoch": 0.12375690607734807, "loss": -0.1636, "grad_norm": 3.878553867340088, "learning_rate": 6.26969696969697e-06, "num_tokens": 2174893.0, "completions/mean_length": 115.625, "completions/min_length": 38.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 59.000003814697266, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.6648431420326233, "rewards/meter/std": 0.3916427791118622, "rewards/count_adherence/mean": 0.8333333730697632, "rewards/count_adherence/std": 0.17817415297031403, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9351570010185242, "rewards/repeat_soft/std": 0.07089769840240479, "rewards/judge_quality/mean": 0.5237500071525574, "rewards/judge_quality/std": 0.24224768579006195, "rewards/total_composite/mean": 0.5185441970825195, "rewards/total_composite/std": 0.26280826330184937, "reward": 0.5185441970825195, "reward_std": 0.26280826330184937, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1412878930568695, "sampling/sampling_logp_difference/max": 2.252089023590088, "sampling/importance_sampling_ratio/min": 0.10517927259206772, "sampling/importance_sampling_ratio/mean": 1.015083909034729, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8335136622190475, "clip_ratio/low_mean": 0.02855603490024805, "clip_ratio/low_min": 0.02855603490024805, "clip_ratio/high_mean": 0.08079096302390099, "clip_ratio/high_max": 0.08079096302390099, "clip_ratio/region_mean": 0.10934699792414904, "reward_total_mean": 0.5185441970825195, "reward_meter_mean": 0.6648431420326233, "reward_meter_std": 0.3916427791118622, "reward_count_adherence_mean": 0.8333333730697632, "reward_count_adherence_std": 0.17817415297031403, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9351570010185242, "reward_repeat_soft_std": 0.07089769840240479, "reward_judge_quality_mean": 0.5237500071525574, "reward_judge_quality_std": 0.24224768579006195, "reward_total_composite_mean": 0.5185441970825195, "reward_total_composite_std": 0.26280826330184937} {"timestamp_utc": "2026-04-13T10:24:15Z", "mode": "train", "global_step": 1233, "epoch": 0.12385735811150175, "loss": -0.1058, "grad_norm": 2.93407940864563, "learning_rate": 6.266666666666668e-06, "num_tokens": 2176433.0, "completions/mean_length": 97.5, "completions/min_length": 28.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 38.28571701049805, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.6622532606124878, "rewards/meter/std": 0.3364217281341553, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9898523092269897, "rewards/repeat_soft/std": 0.019080858677625656, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.28965190052986145, "rewards/total_composite/mean": 0.5197000503540039, "rewards/total_composite/std": 0.2545000910758972, "reward": 0.5197000503540039, "reward_std": 0.2545000910758972, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13664236664772034, "sampling/sampling_logp_difference/max": 1.1050457954406738, "sampling/importance_sampling_ratio/min": 0.3311957120895386, "sampling/importance_sampling_ratio/mean": 1.0266621112823486, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.853924110531807, "clip_ratio/low_mean": 0.04910714365541935, "clip_ratio/low_min": 0.04910714365541935, "clip_ratio/high_mean": 0.10055717173963785, "clip_ratio/high_max": 0.10055717173963785, "clip_ratio/region_mean": 0.1496643153950572, "reward_total_mean": 0.5197000503540039, "reward_meter_mean": 0.6622532606124878, "reward_meter_std": 0.3364217281341553, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9898523092269897, "reward_repeat_soft_std": 0.019080858677625656, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.28965190052986145, "reward_total_composite_mean": 0.5197000503540039, "reward_total_composite_std": 0.2545000910758972} {"timestamp_utc": "2026-04-13T10:24:21Z", "mode": "train", "global_step": 1234, "epoch": 0.12395781014565545, "loss": -0.0091, "grad_norm": 12.671320915222168, "learning_rate": 6.263636363636364e-06, "num_tokens": 2178029.0, "completions/mean_length": 32.5, "completions/min_length": 28.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.5, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.40019360184669495, "rewards/meter/std": 0.38638606667518616, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9779560565948486, "rewards/repeat_soft/std": 0.021496865898370743, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.4300210475921631, "rewards/total_composite/std": 0.08451351523399353, "reward": 0.4300210475921631, "reward_std": 0.08451351523399353, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.136244535446167, "sampling/sampling_logp_difference/max": 1.4354257583618164, "sampling/importance_sampling_ratio/min": 0.23801401257514954, "sampling/importance_sampling_ratio/mean": 1.0039359331130981, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9081961289048195, "clip_ratio/low_mean": 0.0855260866228491, "clip_ratio/low_min": 0.0855260866228491, "clip_ratio/high_mean": 0.04243326187133789, "clip_ratio/high_max": 0.04243326187133789, "clip_ratio/region_mean": 0.127959348494187, "reward_total_mean": 0.4300210475921631, "reward_meter_mean": 0.40019360184669495, "reward_meter_std": 0.38638606667518616, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9779560565948486, "reward_repeat_soft_std": 0.021496865898370743, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.4300210475921631, "reward_total_composite_std": 0.08451351523399353} {"timestamp_utc": "2026-04-13T10:24:28Z", "mode": "train", "global_step": 1235, "epoch": 0.12405826217980914, "loss": 0.0277, "grad_norm": 10.910974502563477, "learning_rate": 6.260606060606062e-06, "num_tokens": 2179750.0, "completions/mean_length": 47.125, "completions/min_length": 41.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.125, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.6514851450920105, "rewards/meter/std": 0.457214891910553, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9426051378250122, "rewards/repeat_soft/std": 0.08489826321601868, "rewards/judge_quality/mean": 0.5774999856948853, "rewards/judge_quality/std": 0.22461079061031342, "rewards/total_composite/mean": 0.5894964933395386, "rewards/total_composite/std": 0.19443294405937195, "reward": 0.5894964933395386, "reward_std": 0.19443294405937195, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15471337735652924, "sampling/sampling_logp_difference/max": 1.7690258026123047, "sampling/importance_sampling_ratio/min": 0.1704990118741989, "sampling/importance_sampling_ratio/mean": 1.0226340293884277, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.998061791062355, "clip_ratio/low_mean": 0.058662546798586845, "clip_ratio/low_min": 0.058662546798586845, "clip_ratio/high_mean": 0.06591849401593208, "clip_ratio/high_max": 0.06591849401593208, "clip_ratio/region_mean": 0.12458104081451893, "reward_total_mean": 0.5894964933395386, "reward_meter_mean": 0.6514851450920105, "reward_meter_std": 0.457214891910553, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9426051378250122, "reward_repeat_soft_std": 0.08489826321601868, "reward_judge_quality_mean": 0.5774999856948853, "reward_judge_quality_std": 0.22461079061031342, "reward_total_composite_mean": 0.5894964933395386, "reward_total_composite_std": 0.19443294405937195} {"timestamp_utc": "2026-04-13T10:24:35Z", "mode": "train", "global_step": 1236, "epoch": 0.12415871421396284, "loss": -0.0314, "grad_norm": 18.540569305419922, "learning_rate": 6.257575757575758e-06, "num_tokens": 2181435.0, "completions/mean_length": 31.625, "completions/min_length": 26.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.625, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.5110172033309937, "rewards/meter/std": 0.3751196563243866, "rewards/count_adherence/mean": 0.7083333730697632, "rewards/count_adherence/std": 0.11785111576318741, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9423044323921204, "rewards/repeat_soft/std": 0.05376303568482399, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.42815661430358887, "rewards/total_composite/std": 0.11032947897911072, "reward": 0.42815661430358887, "reward_std": 0.11032947152853012, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1298311948776245, "sampling/sampling_logp_difference/max": 1.2114909887313843, "sampling/importance_sampling_ratio/min": 0.2977529764175415, "sampling/importance_sampling_ratio/mean": 1.0086385011672974, "sampling/importance_sampling_ratio/max": 1.7687610387802124, "entropy": 0.8549127578735352, "clip_ratio/low_mean": 0.07775939162820578, "clip_ratio/low_min": 0.07775939162820578, "clip_ratio/high_mean": 0.07042957190424204, "clip_ratio/high_max": 0.07042957190424204, "clip_ratio/region_mean": 0.14818896353244781, "reward_total_mean": 0.42815661430358887, "reward_meter_mean": 0.5110172033309937, "reward_meter_std": 0.3751196563243866, "reward_count_adherence_mean": 0.7083333730697632, "reward_count_adherence_std": 0.11785111576318741, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9423044323921204, "reward_repeat_soft_std": 0.05376303568482399, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.42815661430358887, "reward_total_composite_std": 0.11032947897911072} {"timestamp_utc": "2026-04-13T10:24:43Z", "mode": "train", "global_step": 1237, "epoch": 0.12425916624811652, "loss": -0.0864, "grad_norm": 13.851546287536621, "learning_rate": 6.254545454545455e-06, "num_tokens": 2183394.0, "completions/mean_length": 58.875, "completions/min_length": 48.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.875, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.7594131231307983, "rewards/meter/std": 0.28806328773498535, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.15430334210395813, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.985525369644165, "rewards/repeat_soft/std": 0.0346342995762825, "rewards/judge_quality/mean": 0.4025000035762787, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.49545133113861084, "rewards/total_composite/std": 0.0897204726934433, "reward": 0.49545133113861084, "reward_std": 0.0897204726934433, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17434343695640564, "sampling/sampling_logp_difference/max": 2.0161962509155273, "sampling/importance_sampling_ratio/min": 0.1331610232591629, "sampling/importance_sampling_ratio/mean": 1.0120218992233276, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1033561006188393, "clip_ratio/low_mean": 0.07087666913866997, "clip_ratio/low_min": 0.07087666913866997, "clip_ratio/high_mean": 0.07048771437257528, "clip_ratio/high_max": 0.07048771437257528, "clip_ratio/region_mean": 0.14136438351124525, "reward_total_mean": 0.49545133113861084, "reward_meter_mean": 0.7594131231307983, "reward_meter_std": 0.28806328773498535, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.15430334210395813, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.985525369644165, "reward_repeat_soft_std": 0.0346342995762825, "reward_judge_quality_mean": 0.4025000035762787, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.49545133113861084, "reward_total_composite_std": 0.0897204726934433} {"timestamp_utc": "2026-04-13T10:24:49Z", "mode": "train", "global_step": 1238, "epoch": 0.12435961828227021, "loss": 0.018, "grad_norm": 11.206317901611328, "learning_rate": 6.251515151515152e-06, "num_tokens": 2185182.0, "completions/mean_length": 53.5, "completions/min_length": 39.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.5, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.9882075786590576, "rewards/meter/std": 0.007610471919178963, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9839343428611755, "rewards/repeat_soft/std": 0.015634674578905106, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.6221863627433777, "rewards/total_composite/std": 0.009470634162425995, "reward": 0.6221863627433777, "reward_std": 0.009470636025071144, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1383993923664093, "sampling/sampling_logp_difference/max": 1.1137313842773438, "sampling/importance_sampling_ratio/min": 0.32833153009414673, "sampling/importance_sampling_ratio/mean": 1.0288126468658447, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1552726477384567, "clip_ratio/low_mean": 0.06292570754885674, "clip_ratio/low_min": 0.06292570754885674, "clip_ratio/high_mean": 0.06277895253151655, "clip_ratio/high_max": 0.06277895253151655, "clip_ratio/region_mean": 0.1257046600803733, "reward_total_mean": 0.6221863627433777, "reward_meter_mean": 0.9882075786590576, "reward_meter_std": 0.007610471919178963, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9839343428611755, "reward_repeat_soft_std": 0.015634674578905106, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.6221863627433777, "reward_total_composite_std": 0.009470634162425995} {"timestamp_utc": "2026-04-13T10:24:57Z", "mode": "train", "global_step": 1239, "epoch": 0.12446007031642391, "loss": 0.0055, "grad_norm": 9.425826072692871, "learning_rate": 6.248484848484849e-06, "num_tokens": 2187544.0, "completions/mean_length": 96.25, "completions/min_length": 75.0, "completions/max_length": 106.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.25, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.8532570600509644, "rewards/meter/std": 0.23092512786388397, "rewards/count_adherence/mean": 0.6458333730697632, "rewards/count_adherence/std": 0.0589255727827549, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9596713781356812, "rewards/repeat_soft/std": 0.02742459625005722, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.5543482303619385, "rewards/total_composite/std": 0.1316426694393158, "reward": 0.5543482303619385, "reward_std": 0.1316426545381546, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14521437883377075, "sampling/sampling_logp_difference/max": 1.6073007583618164, "sampling/importance_sampling_ratio/min": 0.2171570062637329, "sampling/importance_sampling_ratio/mean": 1.0121631622314453, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0436527580022812, "clip_ratio/low_mean": 0.10271982569247484, "clip_ratio/low_min": 0.10271982569247484, "clip_ratio/high_mean": 0.030097167938947678, "clip_ratio/high_max": 0.030097167938947678, "clip_ratio/region_mean": 0.13281699363142252, "reward_total_mean": 0.5543482303619385, "reward_meter_mean": 0.8532570600509644, "reward_meter_std": 0.23092512786388397, "reward_count_adherence_mean": 0.6458333730697632, "reward_count_adherence_std": 0.0589255727827549, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9596713781356812, "reward_repeat_soft_std": 0.02742459625005722, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.5543482303619385, "reward_total_composite_std": 0.1316426694393158} {"timestamp_utc": "2026-04-13T10:25:03Z", "mode": "train", "global_step": 1240, "epoch": 0.1245605223505776, "loss": -0.0157, "grad_norm": 12.375225067138672, "learning_rate": 6.245454545454545e-06, "num_tokens": 2189245.0, "completions/mean_length": 52.625, "completions/min_length": 44.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.625, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.7396665215492249, "rewards/meter/std": 0.3439599275588989, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9697411060333252, "rewards/repeat_soft/std": 0.02785804308950901, "rewards/judge_quality/mean": 0.4937500059604645, "rewards/judge_quality/std": 0.1728696972131729, "rewards/total_composite/mean": 0.5908524990081787, "rewards/total_composite/std": 0.15924030542373657, "reward": 0.5908524990081787, "reward_std": 0.15924030542373657, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14979474246501923, "sampling/sampling_logp_difference/max": 2.096376419067383, "sampling/importance_sampling_ratio/min": 0.12290095537900925, "sampling/importance_sampling_ratio/mean": 1.023373007774353, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0163159668445587, "clip_ratio/low_mean": 0.047969188541173935, "clip_ratio/low_min": 0.047969188541173935, "clip_ratio/high_mean": 0.07672291994094849, "clip_ratio/high_max": 0.07672291994094849, "clip_ratio/region_mean": 0.12469210848212242, "reward_total_mean": 0.5908524990081787, "reward_meter_mean": 0.7396665215492249, "reward_meter_std": 0.3439599275588989, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9697411060333252, "reward_repeat_soft_std": 0.02785804308950901, "reward_judge_quality_mean": 0.4937500059604645, "reward_judge_quality_std": 0.1728696972131729, "reward_total_composite_mean": 0.5908524990081787, "reward_total_composite_std": 0.15924030542373657} {"timestamp_utc": "2026-04-13T10:25:09Z", "mode": "train", "global_step": 1241, "epoch": 0.1246609743847313, "loss": 0.033, "grad_norm": 13.497233390808105, "learning_rate": 6.2424242424242434e-06, "num_tokens": 2190866.0, "completions/mean_length": 50.625, "completions/min_length": 40.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.625, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9617208242416382, "rewards/meter/std": 0.07272892445325851, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.968368411064148, "rewards/repeat_soft/std": 0.028197133913636208, "rewards/judge_quality/mean": 0.5862499475479126, "rewards/judge_quality/std": 0.19558978080749512, "rewards/total_composite/mean": 0.714325487613678, "rewards/total_composite/std": 0.1335224211215973, "reward": 0.714325487613678, "reward_std": 0.1335224211215973, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1718737632036209, "sampling/sampling_logp_difference/max": 1.2398936748504639, "sampling/importance_sampling_ratio/min": 0.29771336913108826, "sampling/importance_sampling_ratio/mean": 1.018097996711731, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3330445513129234, "clip_ratio/low_mean": 0.0491032125428319, "clip_ratio/low_min": 0.0491032125428319, "clip_ratio/high_mean": 0.07631996087729931, "clip_ratio/high_max": 0.07631996087729931, "clip_ratio/region_mean": 0.1254231734201312, "reward_total_mean": 0.714325487613678, "reward_meter_mean": 0.9617208242416382, "reward_meter_std": 0.07272892445325851, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.968368411064148, "reward_repeat_soft_std": 0.028197133913636208, "reward_judge_quality_mean": 0.5862499475479126, "reward_judge_quality_std": 0.19558978080749512, "reward_total_composite_mean": 0.714325487613678, "reward_total_composite_std": 0.1335224211215973} {"timestamp_utc": "2026-04-13T10:25:20Z", "mode": "train", "global_step": 1242, "epoch": 0.12476142641888498, "loss": -0.1331, "grad_norm": 2.4201745986938477, "learning_rate": 6.23939393939394e-06, "num_tokens": 2192596.0, "completions/mean_length": 109.25, "completions/min_length": 45.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 51.71428680419922, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.8386589288711548, "rewards/meter/std": 0.26981183886528015, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9880883693695068, "rewards/repeat_soft/std": 0.0190111193805933, "rewards/judge_quality/mean": 0.4099999964237213, "rewards/judge_quality/std": 0.15982133150100708, "rewards/total_composite/mean": 0.5404645800590515, "rewards/total_composite/std": 0.2306319624185562, "reward": 0.5404645800590515, "reward_std": 0.2306319624185562, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1571856141090393, "sampling/sampling_logp_difference/max": 1.2705446481704712, "sampling/importance_sampling_ratio/min": 0.28067871928215027, "sampling/importance_sampling_ratio/mean": 1.0289143323898315, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0195204690098763, "clip_ratio/low_mean": 0.01785714365541935, "clip_ratio/low_min": 0.01785714365541935, "clip_ratio/high_mean": 0.10808935575187206, "clip_ratio/high_max": 0.10808935575187206, "clip_ratio/region_mean": 0.1259464994072914, "reward_total_mean": 0.5404645800590515, "reward_meter_mean": 0.8386589288711548, "reward_meter_std": 0.26981183886528015, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9880883693695068, "reward_repeat_soft_std": 0.0190111193805933, "reward_judge_quality_mean": 0.4099999964237213, "reward_judge_quality_std": 0.15982133150100708, "reward_total_composite_mean": 0.5404645800590515, "reward_total_composite_std": 0.2306319624185562} {"timestamp_utc": "2026-04-13T10:25:27Z", "mode": "train", "global_step": 1243, "epoch": 0.12486187845303867, "loss": 0.0496, "grad_norm": 10.702670097351074, "learning_rate": 6.236363636363637e-06, "num_tokens": 2194803.0, "completions/mean_length": 76.875, "completions/min_length": 60.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.875, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.46705207228660583, "rewards/meter/std": 0.3572114109992981, "rewards/count_adherence/mean": 0.6000000238418579, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9794241189956665, "rewards/repeat_soft/std": 0.013340122997760773, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.3944188356399536, "rewards/total_composite/std": 0.09902052581310272, "reward": 0.3944188356399536, "reward_std": 0.09902053326368332, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15322375297546387, "sampling/sampling_logp_difference/max": 2.087918281555176, "sampling/importance_sampling_ratio/min": 0.12394487857818604, "sampling/importance_sampling_ratio/mean": 1.0137479305267334, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9834109246730804, "clip_ratio/low_mean": 0.06600053561851382, "clip_ratio/low_min": 0.06600053561851382, "clip_ratio/high_mean": 0.05654762126505375, "clip_ratio/high_max": 0.05654762126505375, "clip_ratio/region_mean": 0.12254815688356757, "reward_total_mean": 0.3944188356399536, "reward_meter_mean": 0.46705207228660583, "reward_meter_std": 0.3572114109992981, "reward_count_adherence_mean": 0.6000000238418579, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9794241189956665, "reward_repeat_soft_std": 0.013340122997760773, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.3944188356399536, "reward_total_composite_std": 0.09902052581310272} {"timestamp_utc": "2026-04-13T10:25:38Z", "mode": "train", "global_step": 1244, "epoch": 0.12496233048719237, "loss": -0.1396, "grad_norm": 3.0950686931610107, "learning_rate": 6.2333333333333335e-06, "num_tokens": 2196963.0, "completions/mean_length": 135.0, "completions/min_length": 71.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 81.14286041259766, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.8606762886047363, "rewards/meter/std": 0.34499090909957886, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9489689469337463, "rewards/repeat_soft/std": 0.04853681102395058, "rewards/judge_quality/mean": 0.6899999976158142, "rewards/judge_quality/std": 0.32035693526268005, "rewards/total_composite/mean": 0.685646116733551, "rewards/total_composite/std": 0.30944472551345825, "reward": 0.685646116733551, "reward_std": 0.30944475531578064, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14565177261829376, "sampling/sampling_logp_difference/max": 1.917027235031128, "sampling/importance_sampling_ratio/min": 0.14704343676567078, "sampling/importance_sampling_ratio/mean": 1.0036393404006958, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9305947348475456, "clip_ratio/low_mean": 0.0375093650072813, "clip_ratio/low_min": 0.0375093650072813, "clip_ratio/high_mean": 0.06487716641277075, "clip_ratio/high_max": 0.06487716641277075, "clip_ratio/region_mean": 0.10238653142005205, "reward_total_mean": 0.685646116733551, "reward_meter_mean": 0.8606762886047363, "reward_meter_std": 0.34499090909957886, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9489689469337463, "reward_repeat_soft_std": 0.04853681102395058, "reward_judge_quality_mean": 0.6899999976158142, "reward_judge_quality_std": 0.32035693526268005, "reward_total_composite_mean": 0.685646116733551, "reward_total_composite_std": 0.30944472551345825} {"timestamp_utc": "2026-04-13T10:25:44Z", "mode": "train", "global_step": 1245, "epoch": 0.12506278252134606, "loss": 0.0049, "grad_norm": 23.15663719177246, "learning_rate": 6.230303030303031e-06, "num_tokens": 2198306.0, "completions/mean_length": 22.875, "completions/min_length": 20.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.875, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9307630062103271, "rewards/meter/std": 0.1662379801273346, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9394959211349487, "rewards/repeat_soft/std": 0.06506536155939102, "rewards/judge_quality/mean": 0.4649999737739563, "rewards/judge_quality/std": 0.1940544992685318, "rewards/total_composite/mean": 0.6307128667831421, "rewards/total_composite/std": 0.14116933941841125, "reward": 0.6307128667831421, "reward_std": 0.14116933941841125, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16882959008216858, "sampling/sampling_logp_difference/max": 1.842193365097046, "sampling/importance_sampling_ratio/min": 0.15846946835517883, "sampling/importance_sampling_ratio/mean": 1.0106979608535767, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7975779548287392, "clip_ratio/low_mean": 0.0898249763995409, "clip_ratio/low_min": 0.0898249763995409, "clip_ratio/high_mean": 0.03217391204088926, "clip_ratio/high_max": 0.03217391204088926, "clip_ratio/region_mean": 0.12199888844043016, "reward_total_mean": 0.6307128667831421, "reward_meter_mean": 0.9307630062103271, "reward_meter_std": 0.1662379801273346, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9394959211349487, "reward_repeat_soft_std": 0.06506536155939102, "reward_judge_quality_mean": 0.4649999737739563, "reward_judge_quality_std": 0.1940544992685318, "reward_total_composite_mean": 0.6307128667831421, "reward_total_composite_std": 0.14116933941841125} {"timestamp_utc": "2026-04-13T10:25:50Z", "mode": "train", "global_step": 1246, "epoch": 0.12516323455549974, "loss": 0.0269, "grad_norm": 13.141584396362305, "learning_rate": 6.227272727272727e-06, "num_tokens": 2199643.0, "completions/mean_length": 23.125, "completions/min_length": 18.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.125, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.986322283744812, "rewards/meter/std": 0.006948336027562618, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9170810580253601, "rewards/repeat_soft/std": 0.06967313587665558, "rewards/judge_quality/mean": 0.42124998569488525, "rewards/judge_quality/std": 0.06998724490404129, "rewards/total_composite/mean": 0.6075693368911743, "rewards/total_composite/std": 0.05287746340036392, "reward": 0.6075693368911743, "reward_std": 0.05287746340036392, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15039460361003876, "sampling/sampling_logp_difference/max": 1.7972307205200195, "sampling/importance_sampling_ratio/min": 0.16575726866722107, "sampling/importance_sampling_ratio/mean": 1.0215646028518677, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0975348427891731, "clip_ratio/low_mean": 0.009999999776482582, "clip_ratio/low_min": 0.009999999776482582, "clip_ratio/high_mean": 0.09229386271908879, "clip_ratio/high_max": 0.09229386271908879, "clip_ratio/region_mean": 0.10229386249557137, "reward_total_mean": 0.6075693368911743, "reward_meter_mean": 0.986322283744812, "reward_meter_std": 0.006948336027562618, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9170810580253601, "reward_repeat_soft_std": 0.06967313587665558, "reward_judge_quality_mean": 0.42124998569488525, "reward_judge_quality_std": 0.06998724490404129, "reward_total_composite_mean": 0.6075693368911743, "reward_total_composite_std": 0.05287746340036392} {"timestamp_utc": "2026-04-13T10:26:02Z", "mode": "train", "global_step": 1247, "epoch": 0.12526368658965345, "loss": -0.1776, "grad_norm": 2.893418550491333, "learning_rate": 6.224242424242425e-06, "num_tokens": 2201604.0, "completions/mean_length": 134.125, "completions/min_length": 65.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 80.14286041259766, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.709868311882019, "rewards/meter/std": 0.27947330474853516, "rewards/count_adherence/mean": 0.84375, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9677321314811707, "rewards/repeat_soft/std": 0.021217970177531242, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.45401251316070557, "rewards/total_composite/std": 0.19332006573677063, "reward": 0.45401251316070557, "reward_std": 0.19332006573677063, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15028485655784607, "sampling/sampling_logp_difference/max": 2.978888988494873, "sampling/importance_sampling_ratio/min": 0.050849296152591705, "sampling/importance_sampling_ratio/mean": 1.010841965675354, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8377620056271553, "clip_ratio/low_mean": 0.03322988003492355, "clip_ratio/low_min": 0.03322988003492355, "clip_ratio/high_mean": 0.09654430113732815, "clip_ratio/high_max": 0.09654430113732815, "clip_ratio/region_mean": 0.1297741811722517, "reward_total_mean": 0.45401251316070557, "reward_meter_mean": 0.709868311882019, "reward_meter_std": 0.27947330474853516, "reward_count_adherence_mean": 0.84375, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9677321314811707, "reward_repeat_soft_std": 0.021217970177531242, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.45401251316070557, "reward_total_composite_std": 0.19332006573677063} {"timestamp_utc": "2026-04-13T10:26:09Z", "mode": "train", "global_step": 1248, "epoch": 0.12536413862380713, "loss": 0.0259, "grad_norm": 10.667177200317383, "learning_rate": 6.221212121212121e-06, "num_tokens": 2203609.0, "completions/mean_length": 80.625, "completions/min_length": 74.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.625, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.7917189598083496, "rewards/meter/std": 0.26461970806121826, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.973153293132782, "rewards/repeat_soft/std": 0.030183982104063034, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5621123313903809, "rewards/total_composite/std": 0.07222156971693039, "reward": 0.5621123313903809, "reward_std": 0.07222156226634979, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1624266654253006, "sampling/sampling_logp_difference/max": 2.63492751121521, "sampling/importance_sampling_ratio/min": 0.07172417640686035, "sampling/importance_sampling_ratio/mean": 1.023848056793213, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1514958143234253, "clip_ratio/low_mean": 0.06772187165915966, "clip_ratio/low_min": 0.06772187165915966, "clip_ratio/high_mean": 0.08548036124557257, "clip_ratio/high_max": 0.08548036124557257, "clip_ratio/region_mean": 0.15320223290473223, "reward_total_mean": 0.5621123313903809, "reward_meter_mean": 0.7917189598083496, "reward_meter_std": 0.26461970806121826, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.973153293132782, "reward_repeat_soft_std": 0.030183982104063034, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5621123313903809, "reward_total_composite_std": 0.07222156971693039} {"timestamp_utc": "2026-04-13T10:26:16Z", "mode": "train", "global_step": 1249, "epoch": 0.1254645906579608, "loss": -0.0403, "grad_norm": 14.570401191711426, "learning_rate": 6.218181818181819e-06, "num_tokens": 2205280.0, "completions/mean_length": 43.875, "completions/min_length": 27.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.875, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.6555452942848206, "rewards/meter/std": 0.30486875772476196, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9634662866592407, "rewards/repeat_soft/std": 0.03781601041555405, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.17136012017726898, "rewards/total_composite/mean": 0.5708943605422974, "rewards/total_composite/std": 0.16406582295894623, "reward": 0.5708943605422974, "reward_std": 0.16406580805778503, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17109832167625427, "sampling/sampling_logp_difference/max": 1.7995681762695312, "sampling/importance_sampling_ratio/min": 0.16537028551101685, "sampling/importance_sampling_ratio/mean": 1.0338964462280273, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2643868625164032, "clip_ratio/low_mean": 0.09901054576039314, "clip_ratio/low_min": 0.09901054576039314, "clip_ratio/high_mean": 0.09047590382397175, "clip_ratio/high_max": 0.09047590382397175, "clip_ratio/region_mean": 0.1894864495843649, "reward_total_mean": 0.5708943605422974, "reward_meter_mean": 0.6555452942848206, "reward_meter_std": 0.30486875772476196, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9634662866592407, "reward_repeat_soft_std": 0.03781601041555405, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.17136012017726898, "reward_total_composite_mean": 0.5708943605422974, "reward_total_composite_std": 0.16406582295894623} {"timestamp_utc": "2026-04-13T10:26:23Z", "mode": "train", "global_step": 1250, "epoch": 0.12556504269211452, "loss": 0.0766, "grad_norm": 17.4160213470459, "learning_rate": 6.215151515151515e-06, "num_tokens": 2206889.0, "completions/mean_length": 42.125, "completions/min_length": 38.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.125, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.8886697888374329, "rewards/meter/std": 0.22303172945976257, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9861406087875366, "rewards/repeat_soft/std": 0.016790125519037247, "rewards/judge_quality/mean": 0.8575000166893005, "rewards/judge_quality/std": 0.176776722073555, "rewards/total_composite/mean": 0.8407308459281921, "rewards/total_composite/std": 0.16276921331882477, "reward": 0.8407308459281921, "reward_std": 0.16276919841766357, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15097838640213013, "sampling/sampling_logp_difference/max": 1.4429893493652344, "sampling/importance_sampling_ratio/min": 0.2362205684185028, "sampling/importance_sampling_ratio/mean": 1.0155298709869385, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9842656850814819, "clip_ratio/low_mean": 0.029223228804767132, "clip_ratio/low_min": 0.029223228804767132, "clip_ratio/high_mean": 0.13385965023189783, "clip_ratio/high_max": 0.13385965023189783, "clip_ratio/region_mean": 0.16308287903666496, "reward_total_mean": 0.8407308459281921, "reward_meter_mean": 0.8886697888374329, "reward_meter_std": 0.22303172945976257, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9861406087875366, "reward_repeat_soft_std": 0.016790125519037247, "reward_judge_quality_mean": 0.8575000166893005, "reward_judge_quality_std": 0.176776722073555, "reward_total_composite_mean": 0.8407308459281921, "reward_total_composite_std": 0.16276921331882477} {"timestamp_utc": "2026-04-13T10:27:16Z", "mode": "eval", "global_step": 1250, "epoch": 0.12556504269211452, "eval_loss": NaN, "eval_runtime": 53.0271, "eval_samples_per_second": 1.509, "eval_steps_per_second": 0.189, "eval_num_tokens": 2206889.0, "eval_completions/mean_length": 79.4625, "eval_completions/min_length": 30.6, "eval_completions/max_length": 223.3, "eval_completions/clipped_ratio": 0.0375, "eval_completions/mean_terminated_length": 62.56071548461914, "eval_completions/min_terminated_length": 30.6, "eval_completions/max_terminated_length": 97.8, "eval_rewards/meter/mean": 0.7589913964271545, "eval_rewards/meter/std": 0.2847945909947157, "eval_rewards/count_adherence/mean": 0.8977083384990692, "eval_rewards/count_adherence/std": 0.12962410226464272, "eval_rewards/hard_gate/mean": 0.9625, "eval_rewards/hard_gate/std": 0.10606601536273956, "eval_rewards/repeat_soft/mean": 0.955734246969223, "eval_rewards/repeat_soft/std": 0.04282893724739552, "eval_rewards/judge_quality/mean": 0.47662500441074374, "eval_rewards/judge_quality/std": 0.16404289174824954, "eval_rewards/total_composite/mean": 0.5555821835994721, "eval_rewards/total_composite/std": 0.16687288582324983, "eval_reward": 0.5555821835994721, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.08312602117657661, "eval_sampling/sampling_logp_difference/max": 0.983290719985962, "eval_sampling/importance_sampling_ratio/min": 0.38558162450790406, "eval_sampling/importance_sampling_ratio/mean": 1.0188080310821532, "eval_sampling/importance_sampling_ratio/max": 1.5668889284133911, "eval_entropy": 0.9307186782360077, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5555821835994721, "eval_reward_meter_mean": 0.7589913964271545, "eval_reward_meter_std": 0.2847945909947157, "eval_reward_count_adherence_mean": 0.8977083384990692, "eval_reward_count_adherence_std": 0.12962410226464272, "eval_reward_hard_gate_mean": 0.9625, "eval_reward_hard_gate_std": 0.10606601536273956, "eval_reward_repeat_soft_mean": 0.955734246969223, "eval_reward_repeat_soft_std": 0.04282893724739552, "eval_reward_judge_quality_mean": 0.47662500441074374, "eval_reward_judge_quality_std": 0.16404289174824954, "eval_reward_total_composite_mean": 0.5555821835994721, "eval_reward_total_composite_std": 0.16687288582324983} {"timestamp_utc": "2026-04-13T10:27:31Z", "mode": "train", "global_step": 1251, "epoch": 0.1256654947262682, "loss": -0.1316, "grad_norm": 3.2266314029693604, "learning_rate": 6.212121212121213e-06, "num_tokens": 2208761.0, "completions/mean_length": 129.0, "completions/min_length": 59.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 74.28572082519531, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.47363999485969543, "rewards/meter/std": 0.3648098111152649, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9949181079864502, "rewards/repeat_soft/std": 0.003754963865503669, "rewards/judge_quality/mean": 0.48624998331069946, "rewards/judge_quality/std": 0.23046152293682098, "rewards/total_composite/mean": 0.4828604459762573, "rewards/total_composite/std": 0.24279141426086426, "reward": 0.4828604459762573, "reward_std": 0.24279142916202545, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.155914306640625, "sampling/sampling_logp_difference/max": 1.3624074459075928, "sampling/importance_sampling_ratio/min": 0.25604361295700073, "sampling/importance_sampling_ratio/mean": 1.018334984779358, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0389277786016464, "clip_ratio/low_mean": 0.05251501500606537, "clip_ratio/low_min": 0.05251501500606537, "clip_ratio/high_mean": 0.08952989056706429, "clip_ratio/high_max": 0.08952989056706429, "clip_ratio/region_mean": 0.14204490557312965, "reward_total_mean": 0.4828604459762573, "reward_meter_mean": 0.47363999485969543, "reward_meter_std": 0.3648098111152649, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9949181079864502, "reward_repeat_soft_std": 0.003754963865503669, "reward_judge_quality_mean": 0.48624998331069946, "reward_judge_quality_std": 0.23046152293682098, "reward_total_composite_mean": 0.4828604459762573, "reward_total_composite_std": 0.24279141426086426} {"timestamp_utc": "2026-04-13T10:27:42Z", "mode": "train", "global_step": 1252, "epoch": 0.1257659467604219, "loss": -0.0997, "grad_norm": 3.2271807193756104, "learning_rate": 6.209090909090909e-06, "num_tokens": 2210316.0, "completions/mean_length": 96.375, "completions/min_length": 33.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 37.0, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.7558550834655762, "rewards/meter/std": 0.3269907534122467, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9275693893432617, "rewards/repeat_soft/std": 0.05068346485495567, "rewards/judge_quality/mean": 0.38624998927116394, "rewards/judge_quality/std": 0.15665589272975922, "rewards/total_composite/mean": 0.4778561294078827, "rewards/total_composite/std": 0.21120095252990723, "reward": 0.4778561294078827, "reward_std": 0.21120095252990723, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13341522216796875, "sampling/sampling_logp_difference/max": 1.2374773025512695, "sampling/importance_sampling_ratio/min": 0.2901151776313782, "sampling/importance_sampling_ratio/mean": 1.0314818620681763, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7292150855064392, "clip_ratio/low_mean": 0.0399305559694767, "clip_ratio/low_min": 0.0399305559694767, "clip_ratio/high_mean": 0.055104373721405864, "clip_ratio/high_max": 0.055104373721405864, "clip_ratio/region_mean": 0.09503492969088256, "reward_total_mean": 0.4778561294078827, "reward_meter_mean": 0.7558550834655762, "reward_meter_std": 0.3269907534122467, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9275693893432617, "reward_repeat_soft_std": 0.05068346485495567, "reward_judge_quality_mean": 0.38624998927116394, "reward_judge_quality_std": 0.15665589272975922, "reward_total_composite_mean": 0.4778561294078827, "reward_total_composite_std": 0.21120095252990723} {"timestamp_utc": "2026-04-13T10:27:50Z", "mode": "train", "global_step": 1253, "epoch": 0.1258663987945756, "loss": 0.0138, "grad_norm": 10.025697708129883, "learning_rate": 6.206060606060606e-06, "num_tokens": 2212242.0, "completions/mean_length": 65.75, "completions/min_length": 58.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.75, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.6598318815231323, "rewards/meter/std": 0.27221354842185974, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9810633659362793, "rewards/repeat_soft/std": 0.01688402332365513, "rewards/judge_quality/mean": 0.6825000047683716, "rewards/judge_quality/std": 0.23260945081710815, "rewards/total_composite/mean": 0.6451053619384766, "rewards/total_composite/std": 0.15650221705436707, "reward": 0.6451053619384766, "reward_std": 0.15650221705436707, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1306808590888977, "sampling/sampling_logp_difference/max": 2.628945827484131, "sampling/importance_sampling_ratio/min": 0.07215448468923569, "sampling/importance_sampling_ratio/mean": 0.9933652877807617, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5045273192226887, "clip_ratio/low_mean": 0.05152179766446352, "clip_ratio/low_min": 0.05152179766446352, "clip_ratio/high_mean": 0.06639338470995426, "clip_ratio/high_max": 0.06639338470995426, "clip_ratio/region_mean": 0.11791518237441778, "reward_total_mean": 0.6451053619384766, "reward_meter_mean": 0.6598318815231323, "reward_meter_std": 0.27221354842185974, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9810633659362793, "reward_repeat_soft_std": 0.01688402332365513, "reward_judge_quality_mean": 0.6825000047683716, "reward_judge_quality_std": 0.23260945081710815, "reward_total_composite_mean": 0.6451053619384766, "reward_total_composite_std": 0.15650221705436707} {"timestamp_utc": "2026-04-13T10:28:01Z", "mode": "train", "global_step": 1254, "epoch": 0.12596685082872927, "loss": -0.1015, "grad_norm": 3.218287706375122, "learning_rate": 6.203030303030304e-06, "num_tokens": 2213875.0, "completions/mean_length": 111.125, "completions/min_length": 48.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 53.85714340209961, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.7664738297462463, "rewards/meter/std": 0.2953992187976837, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9914422035217285, "rewards/repeat_soft/std": 0.011019350029528141, "rewards/judge_quality/mean": 0.5400000214576721, "rewards/judge_quality/std": 0.2958281636238098, "rewards/total_composite/mean": 0.5712922811508179, "rewards/total_composite/std": 0.283235102891922, "reward": 0.5712922811508179, "reward_std": 0.283235102891922, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.155284583568573, "sampling/sampling_logp_difference/max": 1.6692895889282227, "sampling/importance_sampling_ratio/min": 0.18838085234165192, "sampling/importance_sampling_ratio/mean": 1.0308421850204468, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1521489322185516, "clip_ratio/low_mean": 0.035098522901535034, "clip_ratio/low_min": 0.035098522901535034, "clip_ratio/high_mean": 0.0953163430094719, "clip_ratio/high_max": 0.0953163430094719, "clip_ratio/region_mean": 0.13041486591100693, "reward_total_mean": 0.5712922811508179, "reward_meter_mean": 0.7664738297462463, "reward_meter_std": 0.2953992187976837, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9914422035217285, "reward_repeat_soft_std": 0.011019350029528141, "reward_judge_quality_mean": 0.5400000214576721, "reward_judge_quality_std": 0.2958281636238098, "reward_total_composite_mean": 0.5712922811508179, "reward_total_composite_std": 0.283235102891922} {"timestamp_utc": "2026-04-13T10:28:08Z", "mode": "train", "global_step": 1255, "epoch": 0.12606730286288298, "loss": 0.0348, "grad_norm": 13.712302207946777, "learning_rate": 6.200000000000001e-06, "num_tokens": 2215651.0, "completions/mean_length": 50.0, "completions/min_length": 44.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.0, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.7190017700195312, "rewards/meter/std": 0.4052329659461975, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.2357022762298584, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9594361186027527, "rewards/repeat_soft/std": 0.03879115730524063, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.11667262762784958, "rewards/total_composite/mean": 0.5490782856941223, "rewards/total_composite/std": 0.14828507602214813, "reward": 0.5490782856941223, "reward_std": 0.14828507602214813, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15267081558704376, "sampling/sampling_logp_difference/max": 1.351280927658081, "sampling/importance_sampling_ratio/min": 0.25890839099884033, "sampling/importance_sampling_ratio/mean": 1.0056182146072388, "sampling/importance_sampling_ratio/max": 1.9183979034423828, "entropy": 0.8657406121492386, "clip_ratio/low_mean": 0.06455399002879858, "clip_ratio/low_min": 0.06455399002879858, "clip_ratio/high_mean": 0.1000808821991086, "clip_ratio/high_max": 0.1000808821991086, "clip_ratio/region_mean": 0.16463487222790718, "reward_total_mean": 0.5490782856941223, "reward_meter_mean": 0.7190017700195312, "reward_meter_std": 0.4052329659461975, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.2357022762298584, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9594361186027527, "reward_repeat_soft_std": 0.03879115730524063, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.11667262762784958, "reward_total_composite_mean": 0.5490782856941223, "reward_total_composite_std": 0.14828507602214813} {"timestamp_utc": "2026-04-13T10:28:14Z", "mode": "train", "global_step": 1256, "epoch": 0.12616775489703666, "loss": 0.0037, "grad_norm": 11.209565162658691, "learning_rate": 6.196969696969698e-06, "num_tokens": 2217223.0, "completions/mean_length": 39.5, "completions/min_length": 31.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.5, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.6701686382293701, "rewards/meter/std": 0.3509586751461029, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9642779231071472, "rewards/repeat_soft/std": 0.04290106147527695, "rewards/judge_quality/mean": 0.4724999964237213, "rewards/judge_quality/std": 0.1011011004447937, "rewards/total_composite/mean": 0.49856626987457275, "rewards/total_composite/std": 0.2385425567626953, "reward": 0.49856626987457275, "reward_std": 0.2385425567626953, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15506722033023834, "sampling/sampling_logp_difference/max": 1.6867170333862305, "sampling/importance_sampling_ratio/min": 0.18512627482414246, "sampling/importance_sampling_ratio/mean": 1.0095295906066895, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0518463253974915, "clip_ratio/low_mean": 0.07911140471696854, "clip_ratio/low_min": 0.07911140471696854, "clip_ratio/high_mean": 0.08315983507782221, "clip_ratio/high_max": 0.08315983507782221, "clip_ratio/region_mean": 0.16227123979479074, "reward_total_mean": 0.49856626987457275, "reward_meter_mean": 0.6701686382293701, "reward_meter_std": 0.3509586751461029, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9642779231071472, "reward_repeat_soft_std": 0.04290106147527695, "reward_judge_quality_mean": 0.4724999964237213, "reward_judge_quality_std": 0.1011011004447937, "reward_total_composite_mean": 0.49856626987457275, "reward_total_composite_std": 0.2385425567626953} {"timestamp_utc": "2026-04-13T10:28:21Z", "mode": "train", "global_step": 1257, "epoch": 0.12626820693119037, "loss": 0.1837, "grad_norm": 17.26266860961914, "learning_rate": 6.1939393939393944e-06, "num_tokens": 2218717.0, "completions/mean_length": 36.75, "completions/min_length": 26.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.75, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.7167240977287292, "rewards/meter/std": 0.43885913491249084, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.4629100561141968, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9467388391494751, "rewards/repeat_soft/std": 0.0382780097424984, "rewards/judge_quality/mean": 0.4424999952316284, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.4993710517883301, "rewards/total_composite/std": 0.14655928313732147, "reward": 0.4993710517883301, "reward_std": 0.14655928313732147, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1434856504201889, "sampling/sampling_logp_difference/max": 1.4102238416671753, "sampling/importance_sampling_ratio/min": 0.24408864974975586, "sampling/importance_sampling_ratio/mean": 1.023720383644104, "sampling/importance_sampling_ratio/max": 1.8100582361221313, "entropy": 1.0456532463431358, "clip_ratio/low_mean": 0.06550586502999067, "clip_ratio/low_min": 0.06550586502999067, "clip_ratio/high_mean": 0.07707366440445185, "clip_ratio/high_max": 0.07707366440445185, "clip_ratio/region_mean": 0.14257952943444252, "reward_total_mean": 0.4993710517883301, "reward_meter_mean": 0.7167240977287292, "reward_meter_std": 0.43885913491249084, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.4629100561141968, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9467388391494751, "reward_repeat_soft_std": 0.0382780097424984, "reward_judge_quality_mean": 0.4424999952316284, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.4993710517883301, "reward_total_composite_std": 0.14655928313732147} {"timestamp_utc": "2026-04-13T10:28:27Z", "mode": "train", "global_step": 1258, "epoch": 0.12636865896534405, "loss": 0.0344, "grad_norm": 27.869922637939453, "learning_rate": 6.190909090909092e-06, "num_tokens": 2220012.0, "completions/mean_length": 14.875, "completions/min_length": 14.0, "completions/max_length": 17.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 14.875, "completions/min_terminated_length": 14.0, "completions/max_terminated_length": 17.0, "rewards/meter/mean": 0.7957995533943176, "rewards/meter/std": 0.22487275302410126, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5616282820701599, "rewards/total_composite/std": 0.061390265822410583, "reward": 0.5616282820701599, "reward_std": 0.06139027327299118, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07037394493818283, "sampling/sampling_logp_difference/max": 1.1815903186798096, "sampling/importance_sampling_ratio/min": 0.30679044127464294, "sampling/importance_sampling_ratio/mean": 0.9939095973968506, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3817711900919676, "clip_ratio/low_mean": 0.056197479367256165, "clip_ratio/low_min": 0.056197479367256165, "clip_ratio/high_mean": 0.03511904925107956, "clip_ratio/high_max": 0.03511904925107956, "clip_ratio/region_mean": 0.09131652861833572, "reward_total_mean": 0.5616282820701599, "reward_meter_mean": 0.7957995533943176, "reward_meter_std": 0.22487275302410126, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5616282820701599, "reward_total_composite_std": 0.061390265822410583} {"timestamp_utc": "2026-04-13T10:28:40Z", "mode": "train", "global_step": 1259, "epoch": 0.12646911099949773, "loss": 0.0345, "grad_norm": 10.268722534179688, "learning_rate": 6.187878787878788e-06, "num_tokens": 2222222.0, "completions/mean_length": 96.25, "completions/min_length": 78.0, "completions/max_length": 125.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.25, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.8196989893913269, "rewards/meter/std": 0.2490110844373703, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.13363061845302582, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9899951219558716, "rewards/repeat_soft/std": 0.008460160344839096, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5472770929336548, "rewards/total_composite/std": 0.05546480044722557, "reward": 0.5472770929336548, "reward_std": 0.05546480044722557, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16437067091464996, "sampling/sampling_logp_difference/max": 1.327692985534668, "sampling/importance_sampling_ratio/min": 0.26508811116218567, "sampling/importance_sampling_ratio/mean": 1.0197168588638306, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1947279572486877, "clip_ratio/low_mean": 0.05987168662250042, "clip_ratio/low_min": 0.05987168662250042, "clip_ratio/high_mean": 0.09439416974782944, "clip_ratio/high_max": 0.09439416974782944, "clip_ratio/region_mean": 0.15426585637032986, "reward_total_mean": 0.5472770929336548, "reward_meter_mean": 0.8196989893913269, "reward_meter_std": 0.2490110844373703, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.13363061845302582, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9899951219558716, "reward_repeat_soft_std": 0.008460160344839096, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5472770929336548, "reward_total_composite_std": 0.05546480044722557} {"timestamp_utc": "2026-04-13T10:28:48Z", "mode": "train", "global_step": 1260, "epoch": 0.12656956303365144, "loss": 0.008, "grad_norm": 8.31602954864502, "learning_rate": 6.184848484848485e-06, "num_tokens": 2224439.0, "completions/mean_length": 113.125, "completions/min_length": 94.0, "completions/max_length": 125.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 113.125, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.8391289710998535, "rewards/meter/std": 0.24904930591583252, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9839187264442444, "rewards/repeat_soft/std": 0.014312672428786755, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.6189167499542236, "rewards/total_composite/std": 0.1338718831539154, "reward": 0.6189167499542236, "reward_std": 0.1338718980550766, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1532590389251709, "sampling/sampling_logp_difference/max": 1.7896296977996826, "sampling/importance_sampling_ratio/min": 0.1670220047235489, "sampling/importance_sampling_ratio/mean": 1.0085344314575195, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1714848205447197, "clip_ratio/low_mean": 0.07738458458334208, "clip_ratio/low_min": 0.07738458458334208, "clip_ratio/high_mean": 0.052012067288160324, "clip_ratio/high_max": 0.052012067288160324, "clip_ratio/region_mean": 0.1293966518715024, "reward_total_mean": 0.6189167499542236, "reward_meter_mean": 0.8391289710998535, "reward_meter_std": 0.24904930591583252, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9839187264442444, "reward_repeat_soft_std": 0.014312672428786755, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.6189167499542236, "reward_total_composite_std": 0.1338718831539154} {"timestamp_utc": "2026-04-13T10:29:00Z", "mode": "train", "global_step": 1261, "epoch": 0.12667001506780512, "loss": -0.1167, "grad_norm": 2.189189910888672, "learning_rate": 6.181818181818182e-06, "num_tokens": 2226139.0, "completions/mean_length": 97.5, "completions/min_length": 32.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 38.28571701049805, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.8494099378585815, "rewards/meter/std": 0.3407638669013977, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9858989715576172, "rewards/repeat_soft/std": 0.008430222980678082, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.13274572789669037, "rewards/total_composite/mean": 0.5380397439002991, "rewards/total_composite/std": 0.21747693419456482, "reward": 0.5380397439002991, "reward_std": 0.21747693419456482, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19998954236507416, "sampling/sampling_logp_difference/max": 3.6249895095825195, "sampling/importance_sampling_ratio/min": 0.02664937637746334, "sampling/importance_sampling_ratio/mean": 1.004363775253296, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8866943642497063, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.153715162537992, "clip_ratio/high_max": 0.153715162537992, "clip_ratio/region_mean": 0.153715162537992, "reward_total_mean": 0.5380397439002991, "reward_meter_mean": 0.8494099378585815, "reward_meter_std": 0.3407638669013977, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9858989715576172, "reward_repeat_soft_std": 0.008430222980678082, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.13274572789669037, "reward_total_composite_mean": 0.5380397439002991, "reward_total_composite_std": 0.21747693419456482} {"timestamp_utc": "2026-04-13T10:29:11Z", "mode": "train", "global_step": 1262, "epoch": 0.12677046710195883, "loss": 0.0258, "grad_norm": 5.586556434631348, "learning_rate": 6.17878787878788e-06, "num_tokens": 2227557.0, "completions/mean_length": 89.25, "completions/min_length": 18.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 28.85714340209961, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.7877958416938782, "rewards/meter/std": 0.31315621733665466, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.4629100561141968, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9601781368255615, "rewards/repeat_soft/std": 0.019703209400177002, "rewards/judge_quality/mean": 0.7487499713897705, "rewards/judge_quality/std": 0.332154780626297, "rewards/total_composite/mean": 0.7074991464614868, "rewards/total_composite/std": 0.34219247102737427, "reward": 0.7074991464614868, "reward_std": 0.34219247102737427, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19931429624557495, "sampling/sampling_logp_difference/max": 3.077780246734619, "sampling/importance_sampling_ratio/min": 0.04606138914823532, "sampling/importance_sampling_ratio/mean": 1.0298024415969849, "sampling/importance_sampling_ratio/max": 1.999465823173523, "entropy": 0.8831368088722229, "clip_ratio/low_mean": 0.023333333432674408, "clip_ratio/low_min": 0.023333333432674408, "clip_ratio/high_mean": 0.10578703787177801, "clip_ratio/high_max": 0.10578703787177801, "clip_ratio/region_mean": 0.12912037130445242, "reward_total_mean": 0.7074991464614868, "reward_meter_mean": 0.7877958416938782, "reward_meter_std": 0.31315621733665466, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.4629100561141968, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9601781368255615, "reward_repeat_soft_std": 0.019703209400177002, "reward_judge_quality_mean": 0.7487499713897705, "reward_judge_quality_std": 0.332154780626297, "reward_total_composite_mean": 0.7074991464614868, "reward_total_composite_std": 0.34219247102737427} {"timestamp_utc": "2026-04-13T10:29:22Z", "mode": "train", "global_step": 1263, "epoch": 0.1268709191361125, "loss": -0.1757, "grad_norm": 3.078822612762451, "learning_rate": 6.175757575757576e-06, "num_tokens": 2229732.0, "completions/mean_length": 152.875, "completions/min_length": 73.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 101.5714340209961, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 139.0, "rewards/meter/mean": 0.8817800283432007, "rewards/meter/std": 0.19807694852352142, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9752034544944763, "rewards/repeat_soft/std": 0.01988898403942585, "rewards/judge_quality/mean": 0.47749999165534973, "rewards/judge_quality/std": 0.22839504480361938, "rewards/total_composite/mean": 0.5431589484214783, "rewards/total_composite/std": 0.2479376643896103, "reward": 0.5431589484214783, "reward_std": 0.2479376643896103, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1447705328464508, "sampling/sampling_logp_difference/max": 1.4552173614501953, "sampling/importance_sampling_ratio/min": 0.23334965109825134, "sampling/importance_sampling_ratio/mean": 1.0285652875900269, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.032411813735962, "clip_ratio/low_mean": 0.015287769958376884, "clip_ratio/low_min": 0.015287769958376884, "clip_ratio/high_mean": 0.09570078644901514, "clip_ratio/high_max": 0.09570078644901514, "clip_ratio/region_mean": 0.11098855640739202, "reward_total_mean": 0.5431589484214783, "reward_meter_mean": 0.8817800283432007, "reward_meter_std": 0.19807694852352142, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9752034544944763, "reward_repeat_soft_std": 0.01988898403942585, "reward_judge_quality_mean": 0.47749999165534973, "reward_judge_quality_std": 0.22839504480361938, "reward_total_composite_mean": 0.5431589484214783, "reward_total_composite_std": 0.2479376643896103} {"timestamp_utc": "2026-04-13T10:29:29Z", "mode": "train", "global_step": 1264, "epoch": 0.1269713711702662, "loss": 0.0077, "grad_norm": 7.699588298797607, "learning_rate": 6.1727272727272735e-06, "num_tokens": 2231561.0, "completions/mean_length": 69.625, "completions/min_length": 63.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.625, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9858421087265015, "rewards/meter/std": 0.016840020194649696, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8454391360282898, "rewards/repeat_soft/std": 0.07404352724552155, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.6362375020980835, "rewards/total_composite/std": 0.11040402948856354, "reward": 0.6362375020980835, "reward_std": 0.11040402948856354, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11907019466161728, "sampling/sampling_logp_difference/max": 1.2024345397949219, "sampling/importance_sampling_ratio/min": 0.3004618287086487, "sampling/importance_sampling_ratio/mean": 1.007966160774231, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8413858339190483, "clip_ratio/low_mean": 0.08165121171623468, "clip_ratio/low_min": 0.08165121171623468, "clip_ratio/high_mean": 0.005281690042465925, "clip_ratio/high_max": 0.005281690042465925, "clip_ratio/region_mean": 0.08693290175870061, "reward_total_mean": 0.6362375020980835, "reward_meter_mean": 0.9858421087265015, "reward_meter_std": 0.016840020194649696, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8454391360282898, "reward_repeat_soft_std": 0.07404352724552155, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.6362375020980835, "reward_total_composite_std": 0.11040402948856354} {"timestamp_utc": "2026-04-13T10:29:43Z", "mode": "train", "global_step": 1265, "epoch": 0.1270718232044199, "loss": 0.042, "grad_norm": 8.433825492858887, "learning_rate": 6.16969696969697e-06, "num_tokens": 2233847.0, "completions/mean_length": 114.75, "completions/min_length": 95.0, "completions/max_length": 135.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 114.75, "completions/min_terminated_length": 95.0, "completions/max_terminated_length": 135.0, "rewards/meter/mean": 0.8529812097549438, "rewards/meter/std": 0.3162272274494171, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9877867698669434, "rewards/repeat_soft/std": 0.011356890201568604, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.5834782123565674, "rewards/total_composite/std": 0.10660649836063385, "reward": 0.5834782123565674, "reward_std": 0.10660647600889206, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1506912261247635, "sampling/sampling_logp_difference/max": 2.9099059104919434, "sampling/importance_sampling_ratio/min": 0.05448085442185402, "sampling/importance_sampling_ratio/mean": 1.014310359954834, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1703742891550064, "clip_ratio/low_mean": 0.05421730037778616, "clip_ratio/low_min": 0.05421730037778616, "clip_ratio/high_mean": 0.08516057766973972, "clip_ratio/high_max": 0.08516057766973972, "clip_ratio/region_mean": 0.13937787804752588, "reward_total_mean": 0.5834782123565674, "reward_meter_mean": 0.8529812097549438, "reward_meter_std": 0.3162272274494171, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9877867698669434, "reward_repeat_soft_std": 0.011356890201568604, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.5834782123565674, "reward_total_composite_std": 0.10660649836063385} {"timestamp_utc": "2026-04-13T10:29:54Z", "mode": "train", "global_step": 1266, "epoch": 0.12717227523857358, "loss": -0.1018, "grad_norm": 3.282893419265747, "learning_rate": 6.166666666666667e-06, "num_tokens": 2235270.0, "completions/mean_length": 97.875, "completions/min_length": 33.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 38.71428680419922, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.5459794402122498, "rewards/meter/std": 0.4117375612258911, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9514325857162476, "rewards/repeat_soft/std": 0.07520611584186554, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.4297533631324768, "rewards/total_composite/std": 0.2113325297832489, "reward": 0.4297533631324768, "reward_std": 0.2113325297832489, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13173063099384308, "sampling/sampling_logp_difference/max": 1.712148904800415, "sampling/importance_sampling_ratio/min": 0.1804775446653366, "sampling/importance_sampling_ratio/mean": 1.0067899227142334, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.718297965824604, "clip_ratio/low_mean": 0.03996442258358002, "clip_ratio/low_min": 0.03996442258358002, "clip_ratio/high_mean": 0.07270388770848513, "clip_ratio/high_max": 0.07270388770848513, "clip_ratio/region_mean": 0.11266831029206514, "reward_total_mean": 0.4297533631324768, "reward_meter_mean": 0.5459794402122498, "reward_meter_std": 0.4117375612258911, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9514325857162476, "reward_repeat_soft_std": 0.07520611584186554, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.4297533631324768, "reward_total_composite_std": 0.2113325297832489} {"timestamp_utc": "2026-04-13T10:30:00Z", "mode": "train", "global_step": 1267, "epoch": 0.12727272727272726, "loss": 0.0667, "grad_norm": 11.786508560180664, "learning_rate": 6.163636363636364e-06, "num_tokens": 2236947.0, "completions/mean_length": 41.625, "completions/min_length": 20.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.625, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.8319039344787598, "rewards/meter/std": 0.31794923543930054, "rewards/count_adherence/mean": 0.125, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9469939470291138, "rewards/repeat_soft/std": 0.03675156831741333, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.4016599953174591, "rewards/total_composite/std": 0.1257942020893097, "reward": 0.4016599953174591, "reward_std": 0.1257941871881485, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14323072135448456, "sampling/sampling_logp_difference/max": 1.8149604797363281, "sampling/importance_sampling_ratio/min": 0.16284434497356415, "sampling/importance_sampling_ratio/mean": 1.0372194051742554, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1073968783020973, "clip_ratio/low_mean": 0.03768939524888992, "clip_ratio/low_min": 0.03768939524888992, "clip_ratio/high_mean": 0.08446543430909514, "clip_ratio/high_max": 0.08446543430909514, "clip_ratio/region_mean": 0.12215482955798507, "reward_total_mean": 0.4016599953174591, "reward_meter_mean": 0.8319039344787598, "reward_meter_std": 0.31794923543930054, "reward_count_adherence_mean": 0.125, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9469939470291138, "reward_repeat_soft_std": 0.03675156831741333, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.4016599953174591, "reward_total_composite_std": 0.1257942020893097} {"timestamp_utc": "2026-04-13T10:30:12Z", "mode": "train", "global_step": 1268, "epoch": 0.12737317930688097, "loss": -0.0833, "grad_norm": 3.330807685852051, "learning_rate": 6.160606060606062e-06, "num_tokens": 2238493.0, "completions/mean_length": 95.25, "completions/min_length": 29.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 35.71428680419922, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.16060248017311096, "rewards/meter/std": 0.28669473528862, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9736111164093018, "rewards/repeat_soft/std": 0.04457922652363777, "rewards/judge_quality/mean": 0.5900000333786011, "rewards/judge_quality/std": 0.27994900941848755, "rewards/total_composite/mean": 0.36672818660736084, "rewards/total_composite/std": 0.11939622461795807, "reward": 0.36672818660736084, "reward_std": 0.11939621716737747, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15467126667499542, "sampling/sampling_logp_difference/max": 1.3939218521118164, "sampling/importance_sampling_ratio/min": 0.2481003850698471, "sampling/importance_sampling_ratio/mean": 0.9885116815567017, "sampling/importance_sampling_ratio/max": 1.71488618850708, "entropy": 0.6161579042673111, "clip_ratio/low_mean": 0.07780447881668806, "clip_ratio/low_min": 0.07780447881668806, "clip_ratio/high_mean": 0.06103704310953617, "clip_ratio/high_max": 0.06103704310953617, "clip_ratio/region_mean": 0.13884152192622423, "reward_total_mean": 0.36672818660736084, "reward_meter_mean": 0.16060248017311096, "reward_meter_std": 0.28669473528862, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9736111164093018, "reward_repeat_soft_std": 0.04457922652363777, "reward_judge_quality_mean": 0.5900000333786011, "reward_judge_quality_std": 0.27994900941848755, "reward_total_composite_mean": 0.36672818660736084, "reward_total_composite_std": 0.11939622461795807} {"timestamp_utc": "2026-04-13T10:30:18Z", "mode": "train", "global_step": 1269, "epoch": 0.12747363134103465, "loss": -0.0118, "grad_norm": 13.987906455993652, "learning_rate": 6.157575757575758e-06, "num_tokens": 2239957.0, "completions/mean_length": 37.0, "completions/min_length": 32.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.879828155040741, "rewards/meter/std": 0.22661611437797546, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.999053955078125, "rewards/repeat_soft/std": 0.0014005512930452824, "rewards/judge_quality/mean": 0.9237500429153442, "rewards/judge_quality/std": 0.01060659158974886, "rewards/total_composite/mean": 0.8783953189849854, "rewards/total_composite/std": 0.13688848912715912, "reward": 0.8783953189849854, "reward_std": 0.13688848912715912, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14671094715595245, "sampling/sampling_logp_difference/max": 3.0388565063476562, "sampling/importance_sampling_ratio/min": 0.04788962006568909, "sampling/importance_sampling_ratio/mean": 0.988682746887207, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5730679966509342, "clip_ratio/low_mean": 0.007575757801532745, "clip_ratio/low_min": 0.007575757801532745, "clip_ratio/high_mean": 0.12588980235159397, "clip_ratio/high_max": 0.12588980235159397, "clip_ratio/region_mean": 0.13346556015312672, "reward_total_mean": 0.8783953189849854, "reward_meter_mean": 0.879828155040741, "reward_meter_std": 0.22661611437797546, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.999053955078125, "reward_repeat_soft_std": 0.0014005512930452824, "reward_judge_quality_mean": 0.9237500429153442, "reward_judge_quality_std": 0.01060659158974886, "reward_total_composite_mean": 0.8783953189849854, "reward_total_composite_std": 0.13688848912715912} {"timestamp_utc": "2026-04-13T10:30:25Z", "mode": "train", "global_step": 1270, "epoch": 0.12757408337518836, "loss": 0.0392, "grad_norm": 11.309529304504395, "learning_rate": 6.154545454545455e-06, "num_tokens": 2241818.0, "completions/mean_length": 52.625, "completions/min_length": 43.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.625, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9154290556907654, "rewards/meter/std": 0.06503310799598694, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9116422533988953, "rewards/repeat_soft/std": 0.07046398520469666, "rewards/judge_quality/mean": 0.5487500429153442, "rewards/judge_quality/std": 0.22937415540218353, "rewards/total_composite/mean": 0.6622281074523926, "rewards/total_composite/std": 0.13527481257915497, "reward": 0.6622281074523926, "reward_std": 0.13527479767799377, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11792018264532089, "sampling/sampling_logp_difference/max": 1.9887970685958862, "sampling/importance_sampling_ratio/min": 0.1368599534034729, "sampling/importance_sampling_ratio/mean": 1.0024830102920532, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6192738823592663, "clip_ratio/low_mean": 0.06327737215906382, "clip_ratio/low_min": 0.06327737215906382, "clip_ratio/high_mean": 0.025917832739651203, "clip_ratio/high_max": 0.025917832739651203, "clip_ratio/region_mean": 0.08919520489871502, "reward_total_mean": 0.6622281074523926, "reward_meter_mean": 0.9154290556907654, "reward_meter_std": 0.06503310799598694, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9116422533988953, "reward_repeat_soft_std": 0.07046398520469666, "reward_judge_quality_mean": 0.5487500429153442, "reward_judge_quality_std": 0.22937415540218353, "reward_total_composite_mean": 0.6622281074523926, "reward_total_composite_std": 0.13527481257915497} {"timestamp_utc": "2026-04-13T10:30:31Z", "mode": "train", "global_step": 1271, "epoch": 0.12767453540934204, "loss": -0.0079, "grad_norm": 10.42291259765625, "learning_rate": 6.151515151515152e-06, "num_tokens": 2244176.0, "completions/mean_length": 92.75, "completions/min_length": 84.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.75, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.9904111623764038, "rewards/meter/std": 0.008908153511583805, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9086899757385254, "rewards/repeat_soft/std": 0.045052893459796906, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.4309046268463135, "rewards/total_composite/std": 0.18315014243125916, "reward": 0.4309046268463135, "reward_std": 0.18315015733242035, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14469754695892334, "sampling/sampling_logp_difference/max": 3.127166748046875, "sampling/importance_sampling_ratio/min": 0.04384183511137962, "sampling/importance_sampling_ratio/mean": 1.0195159912109375, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0048793256282806, "clip_ratio/low_mean": 0.03726926166564226, "clip_ratio/low_min": 0.03726926166564226, "clip_ratio/high_mean": 0.07609590562060475, "clip_ratio/high_max": 0.07609590562060475, "clip_ratio/region_mean": 0.11336516728624701, "reward_total_mean": 0.4309046268463135, "reward_meter_mean": 0.9904111623764038, "reward_meter_std": 0.008908153511583805, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9086899757385254, "reward_repeat_soft_std": 0.045052893459796906, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.4309046268463135, "reward_total_composite_std": 0.18315014243125916} {"timestamp_utc": "2026-04-13T10:30:38Z", "mode": "train", "global_step": 1272, "epoch": 0.12777498744349572, "loss": 0.0736, "grad_norm": 9.260095596313477, "learning_rate": 6.148484848484849e-06, "num_tokens": 2246422.0, "completions/mean_length": 97.75, "completions/min_length": 78.0, "completions/max_length": 113.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.75, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 113.0, "rewards/meter/mean": 0.978448748588562, "rewards/meter/std": 0.01660177670419216, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9162514209747314, "rewards/repeat_soft/std": 0.032950934022665024, "rewards/judge_quality/mean": 0.4362499713897705, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.6151977777481079, "rewards/total_composite/std": 0.08692056685686111, "reward": 0.6151977777481079, "reward_std": 0.08692057430744171, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14708460867404938, "sampling/sampling_logp_difference/max": 2.8121001720428467, "sampling/importance_sampling_ratio/min": 0.060078684240579605, "sampling/importance_sampling_ratio/mean": 1.0057787895202637, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8702584803104401, "clip_ratio/low_mean": 0.10920719988644123, "clip_ratio/low_min": 0.10920719988644123, "clip_ratio/high_mean": 0.012500000186264515, "clip_ratio/high_max": 0.012500000186264515, "clip_ratio/region_mean": 0.12170720007270575, "reward_total_mean": 0.6151977777481079, "reward_meter_mean": 0.978448748588562, "reward_meter_std": 0.01660177670419216, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9162514209747314, "reward_repeat_soft_std": 0.032950934022665024, "reward_judge_quality_mean": 0.4362499713897705, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.6151977777481079, "reward_total_composite_std": 0.08692056685686111} {"timestamp_utc": "2026-04-13T10:30:50Z", "mode": "train", "global_step": 1273, "epoch": 0.12787543947764943, "loss": -0.1782, "grad_norm": 2.169473648071289, "learning_rate": 6.1454545454545454e-06, "num_tokens": 2248367.0, "completions/mean_length": 199.125, "completions/min_length": 66.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 94.83333587646484, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.8148660063743591, "rewards/meter/std": 0.23851215839385986, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9603255987167358, "rewards/repeat_soft/std": 0.05261612311005592, "rewards/judge_quality/mean": 0.38999998569488525, "rewards/judge_quality/std": 0.23802761733531952, "rewards/total_composite/mean": 0.4444544315338135, "rewards/total_composite/std": 0.283040851354599, "reward": 0.4444544315338135, "reward_std": 0.283040851354599, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1279352456331253, "sampling/sampling_logp_difference/max": 1.4324660301208496, "sampling/importance_sampling_ratio/min": 0.23871950805187225, "sampling/importance_sampling_ratio/mean": 1.016228437423706, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6070486083626747, "clip_ratio/low_mean": 0.02227722853422165, "clip_ratio/low_min": 0.02227722853422165, "clip_ratio/high_mean": 0.06470211688429117, "clip_ratio/high_max": 0.06470211688429117, "clip_ratio/region_mean": 0.08697934541851282, "reward_total_mean": 0.4444544315338135, "reward_meter_mean": 0.8148660063743591, "reward_meter_std": 0.23851215839385986, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9603255987167358, "reward_repeat_soft_std": 0.05261612311005592, "reward_judge_quality_mean": 0.38999998569488525, "reward_judge_quality_std": 0.23802761733531952, "reward_total_composite_mean": 0.4444544315338135, "reward_total_composite_std": 0.283040851354599} {"timestamp_utc": "2026-04-13T10:30:58Z", "mode": "train", "global_step": 1274, "epoch": 0.1279758915118031, "loss": 0.0791, "grad_norm": 21.68623161315918, "learning_rate": 6.142424242424243e-06, "num_tokens": 2249843.0, "completions/mean_length": 24.5, "completions/min_length": 18.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.5, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.7573328018188477, "rewards/meter/std": 0.32801494002342224, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9540283679962158, "rewards/repeat_soft/std": 0.01626886986196041, "rewards/judge_quality/mean": 0.6737500429153442, "rewards/judge_quality/std": 0.263435423374176, "rewards/total_composite/mean": 0.7008538246154785, "rewards/total_composite/std": 0.22855205833911896, "reward": 0.7008538246154785, "reward_std": 0.22855207324028015, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16179020702838898, "sampling/sampling_logp_difference/max": 1.4722857475280762, "sampling/importance_sampling_ratio/min": 0.22940054535865784, "sampling/importance_sampling_ratio/mean": 1.0096404552459717, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9626705348491669, "clip_ratio/low_mean": 0.05670033674687147, "clip_ratio/low_min": 0.05670033674687147, "clip_ratio/high_mean": 0.09861111082136631, "clip_ratio/high_max": 0.09861111082136631, "clip_ratio/region_mean": 0.15531144756823778, "reward_total_mean": 0.7008538246154785, "reward_meter_mean": 0.7573328018188477, "reward_meter_std": 0.32801494002342224, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9540283679962158, "reward_repeat_soft_std": 0.01626886986196041, "reward_judge_quality_mean": 0.6737500429153442, "reward_judge_quality_std": 0.263435423374176, "reward_total_composite_mean": 0.7008538246154785, "reward_total_composite_std": 0.22855205833911896} {"timestamp_utc": "2026-04-13T10:31:10Z", "mode": "train", "global_step": 1275, "epoch": 0.12807634354595682, "loss": -0.1136, "grad_norm": 3.550029754638672, "learning_rate": 6.139393939393939e-06, "num_tokens": 2251640.0, "completions/mean_length": 108.625, "completions/min_length": 41.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 51.000003814697266, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.8806706070899963, "rewards/meter/std": 0.31563690304756165, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8966113924980164, "rewards/repeat_soft/std": 0.06710071116685867, "rewards/judge_quality/mean": 0.53125, "rewards/judge_quality/std": 0.3508535623550415, "rewards/total_composite/mean": 0.6288752555847168, "rewards/total_composite/std": 0.32005971670150757, "reward": 0.6288752555847168, "reward_std": 0.32005971670150757, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1061878502368927, "sampling/sampling_logp_difference/max": 1.4472990036010742, "sampling/importance_sampling_ratio/min": 0.23520472645759583, "sampling/importance_sampling_ratio/mean": 1.0176953077316284, "sampling/importance_sampling_ratio/max": 1.9982917308807373, "entropy": 0.7043255418539047, "clip_ratio/low_mean": 0.04471670696511865, "clip_ratio/low_min": 0.04471670696511865, "clip_ratio/high_mean": 0.047743055038154125, "clip_ratio/high_max": 0.047743055038154125, "clip_ratio/region_mean": 0.09245976200327277, "reward_total_mean": 0.6288752555847168, "reward_meter_mean": 0.8806706070899963, "reward_meter_std": 0.31563690304756165, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8966113924980164, "reward_repeat_soft_std": 0.06710071116685867, "reward_judge_quality_mean": 0.53125, "reward_judge_quality_std": 0.3508535623550415, "reward_total_composite_mean": 0.6288752555847168, "reward_total_composite_std": 0.32005971670150757} {"timestamp_utc": "2026-04-13T10:31:18Z", "mode": "train", "global_step": 1276, "epoch": 0.1281767955801105, "loss": -0.037, "grad_norm": 8.634065628051758, "learning_rate": 6.136363636363637e-06, "num_tokens": 2254032.0, "completions/mean_length": 91.0, "completions/min_length": 77.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.0, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.9490756392478943, "rewards/meter/std": 0.09576056152582169, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9341107606887817, "rewards/repeat_soft/std": 0.04371308907866478, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.5992085933685303, "rewards/total_composite/std": 0.1208617240190506, "reward": 0.5992085933685303, "reward_std": 0.1208617240190506, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15283799171447754, "sampling/sampling_logp_difference/max": 2.512065887451172, "sampling/importance_sampling_ratio/min": 0.08110052347183228, "sampling/importance_sampling_ratio/mean": 1.011644959449768, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9222359731793404, "clip_ratio/low_mean": 0.10171039402484894, "clip_ratio/low_min": 0.10171039402484894, "clip_ratio/high_mean": 0.01752336509525776, "clip_ratio/high_max": 0.01752336509525776, "clip_ratio/region_mean": 0.1192337591201067, "reward_total_mean": 0.5992085933685303, "reward_meter_mean": 0.9490756392478943, "reward_meter_std": 0.09576056152582169, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9341107606887817, "reward_repeat_soft_std": 0.04371308907866478, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.5992085933685303, "reward_total_composite_std": 0.1208617240190506} {"timestamp_utc": "2026-04-13T10:31:24Z", "mode": "train", "global_step": 1277, "epoch": 0.12827724761426418, "loss": 0.0143, "grad_norm": 18.64078712463379, "learning_rate": 6.133333333333334e-06, "num_tokens": 2255456.0, "completions/mean_length": 29.0, "completions/min_length": 24.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.0, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.13576379418373108, "rewards/meter/std": 0.1197720319032669, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9961292743682861, "rewards/repeat_soft/std": 0.0050612385384738445, "rewards/judge_quality/mean": 0.5012500286102295, "rewards/judge_quality/std": 0.16974246501922607, "rewards/total_composite/mean": 0.40248921513557434, "rewards/total_composite/std": 0.07096210867166519, "reward": 0.40248921513557434, "reward_std": 0.07096210867166519, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14276927709579468, "sampling/sampling_logp_difference/max": 2.2207584381103516, "sampling/importance_sampling_ratio/min": 0.10852676630020142, "sampling/importance_sampling_ratio/mean": 0.9993550777435303, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5751420333981514, "clip_ratio/low_mean": 0.07176368683576584, "clip_ratio/low_min": 0.07176368683576584, "clip_ratio/high_mean": 0.034050178714096546, "clip_ratio/high_max": 0.034050178714096546, "clip_ratio/region_mean": 0.10581386554986238, "reward_total_mean": 0.40248921513557434, "reward_meter_mean": 0.13576379418373108, "reward_meter_std": 0.1197720319032669, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9961292743682861, "reward_repeat_soft_std": 0.0050612385384738445, "reward_judge_quality_mean": 0.5012500286102295, "reward_judge_quality_std": 0.16974246501922607, "reward_total_composite_mean": 0.40248921513557434, "reward_total_composite_std": 0.07096210867166519} {"timestamp_utc": "2026-04-13T10:31:30Z", "mode": "train", "global_step": 1278, "epoch": 0.1283776996484179, "loss": 0.0742, "grad_norm": 16.209163665771484, "learning_rate": 6.130303030303031e-06, "num_tokens": 2257071.0, "completions/mean_length": 46.875, "completions/min_length": 37.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.875, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.9211243987083435, "rewards/meter/std": 0.12842069566249847, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9725692272186279, "rewards/repeat_soft/std": 0.03515557572245598, "rewards/judge_quality/mean": 0.5950000286102295, "rewards/judge_quality/std": 0.24348658323287964, "rewards/total_composite/mean": 0.7098895907402039, "rewards/total_composite/std": 0.1747393012046814, "reward": 0.7098895907402039, "reward_std": 0.1747393161058426, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14765222370624542, "sampling/sampling_logp_difference/max": 2.2170071601867676, "sampling/importance_sampling_ratio/min": 0.10893464833498001, "sampling/importance_sampling_ratio/mean": 1.0145690441131592, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7762776911258698, "clip_ratio/low_mean": 0.06682059494778514, "clip_ratio/low_min": 0.06682059494778514, "clip_ratio/high_mean": 0.055385487619787455, "clip_ratio/high_max": 0.055385487619787455, "clip_ratio/region_mean": 0.1222060825675726, "reward_total_mean": 0.7098895907402039, "reward_meter_mean": 0.9211243987083435, "reward_meter_std": 0.12842069566249847, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9725692272186279, "reward_repeat_soft_std": 0.03515557572245598, "reward_judge_quality_mean": 0.5950000286102295, "reward_judge_quality_std": 0.24348658323287964, "reward_total_composite_mean": 0.7098895907402039, "reward_total_composite_std": 0.1747393012046814} {"timestamp_utc": "2026-04-13T10:31:36Z", "mode": "train", "global_step": 1279, "epoch": 0.12847815168257157, "loss": 0.0277, "grad_norm": 14.2674560546875, "learning_rate": 6.127272727272727e-06, "num_tokens": 2258797.0, "completions/mean_length": 43.75, "completions/min_length": 40.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.75, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.5102277398109436, "rewards/meter/std": 0.2549363672733307, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9901503324508667, "rewards/repeat_soft/std": 0.01757005974650383, "rewards/judge_quality/mean": 0.5099999904632568, "rewards/judge_quality/std": 0.13979578018188477, "rewards/total_composite/mean": 0.5142813920974731, "rewards/total_composite/std": 0.07932232320308685, "reward": 0.5142813920974731, "reward_std": 0.07932231575250626, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11101723462343216, "sampling/sampling_logp_difference/max": 1.4886724948883057, "sampling/importance_sampling_ratio/min": 0.22567205131053925, "sampling/importance_sampling_ratio/mean": 0.9959089159965515, "sampling/importance_sampling_ratio/max": 1.619319200515747, "entropy": 0.5416922643780708, "clip_ratio/low_mean": 0.03985507180914283, "clip_ratio/low_min": 0.03985507180914283, "clip_ratio/high_mean": 0.0694272646214813, "clip_ratio/high_max": 0.0694272646214813, "clip_ratio/region_mean": 0.10928233643062413, "reward_total_mean": 0.5142813920974731, "reward_meter_mean": 0.5102277398109436, "reward_meter_std": 0.2549363672733307, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9901503324508667, "reward_repeat_soft_std": 0.01757005974650383, "reward_judge_quality_mean": 0.5099999904632568, "reward_judge_quality_std": 0.13979578018188477, "reward_total_composite_mean": 0.5142813920974731, "reward_total_composite_std": 0.07932232320308685} {"timestamp_utc": "2026-04-13T10:31:42Z", "mode": "train", "global_step": 1280, "epoch": 0.12857860371672528, "loss": -0.1251, "grad_norm": 17.665630340576172, "learning_rate": 6.1242424242424245e-06, "num_tokens": 2260445.0, "completions/mean_length": 30.0, "completions/min_length": 22.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.0, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.8284243941307068, "rewards/meter/std": 0.318401575088501, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9651641845703125, "rewards/repeat_soft/std": 0.036734238266944885, "rewards/judge_quality/mean": 0.5900000333786011, "rewards/judge_quality/std": 0.22696760296821594, "rewards/total_composite/mean": 0.6755133867263794, "rewards/total_composite/std": 0.1895093470811844, "reward": 0.6755133867263794, "reward_std": 0.1895093470811844, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1389528065919876, "sampling/sampling_logp_difference/max": 1.9500927925109863, "sampling/importance_sampling_ratio/min": 0.1422608643770218, "sampling/importance_sampling_ratio/mean": 1.0018794536590576, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7707121223211288, "clip_ratio/low_mean": 0.09540211223065853, "clip_ratio/low_min": 0.09540211223065853, "clip_ratio/high_mean": 0.052087913267314434, "clip_ratio/high_max": 0.052087913267314434, "clip_ratio/region_mean": 0.14749002549797297, "reward_total_mean": 0.6755133867263794, "reward_meter_mean": 0.8284243941307068, "reward_meter_std": 0.318401575088501, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9651641845703125, "reward_repeat_soft_std": 0.036734238266944885, "reward_judge_quality_mean": 0.5900000333786011, "reward_judge_quality_std": 0.22696760296821594, "reward_total_composite_mean": 0.6755133867263794, "reward_total_composite_std": 0.1895093470811844} {"timestamp_utc": "2026-04-13T10:31:53Z", "mode": "train", "global_step": 1281, "epoch": 0.12867905575087896, "loss": -0.1069, "grad_norm": 3.375562906265259, "learning_rate": 6.121212121212121e-06, "num_tokens": 2262374.0, "completions/mean_length": 148.125, "completions/min_length": 72.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 96.14286041259766, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 139.0, "rewards/meter/mean": 0.5884270071983337, "rewards/meter/std": 0.38525184988975525, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9659799337387085, "rewards/repeat_soft/std": 0.054748762398958206, "rewards/judge_quality/mean": 0.36000001430511475, "rewards/judge_quality/std": 0.140813946723938, "rewards/total_composite/mean": 0.41940566897392273, "rewards/total_composite/std": 0.1943313330411911, "reward": 0.41940566897392273, "reward_std": 0.1943313330411911, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15925894677639008, "sampling/sampling_logp_difference/max": 1.474484920501709, "sampling/importance_sampling_ratio/min": 0.2288966029882431, "sampling/importance_sampling_ratio/mean": 1.011333703994751, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8784201517701149, "clip_ratio/low_mean": 0.04367189295589924, "clip_ratio/low_min": 0.04367189295589924, "clip_ratio/high_mean": 0.08718924224376678, "clip_ratio/high_max": 0.08718924224376678, "clip_ratio/region_mean": 0.13086113519966602, "reward_total_mean": 0.41940566897392273, "reward_meter_mean": 0.5884270071983337, "reward_meter_std": 0.38525184988975525, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9659799337387085, "reward_repeat_soft_std": 0.054748762398958206, "reward_judge_quality_mean": 0.36000001430511475, "reward_judge_quality_std": 0.140813946723938, "reward_total_composite_mean": 0.41940566897392273, "reward_total_composite_std": 0.1943313330411911} {"timestamp_utc": "2026-04-13T10:32:00Z", "mode": "train", "global_step": 1282, "epoch": 0.12877950778503264, "loss": 0.3394, "grad_norm": 14.450453758239746, "learning_rate": 6.118181818181819e-06, "num_tokens": 2263872.0, "completions/mean_length": 35.25, "completions/min_length": 14.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.25, "completions/min_terminated_length": 14.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9769443273544312, "rewards/meter/std": 0.023884175345301628, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.5175492167472839, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624143242835999, "rewards/repeat_soft/std": 0.013508067466318607, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.22720351815223694, "rewards/total_composite/mean": 0.6212908029556274, "rewards/total_composite/std": 0.21466147899627686, "reward": 0.6212908029556274, "reward_std": 0.21466147899627686, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17344297468662262, "sampling/sampling_logp_difference/max": 2.619584321975708, "sampling/importance_sampling_ratio/min": 0.23441235721111298, "sampling/importance_sampling_ratio/mean": 0.9908496737480164, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8512249290943146, "clip_ratio/low_mean": 0.08545100968331099, "clip_ratio/low_min": 0.08545100968331099, "clip_ratio/high_mean": 0.05839285720139742, "clip_ratio/high_max": 0.05839285720139742, "clip_ratio/region_mean": 0.1438438668847084, "reward_total_mean": 0.6212908029556274, "reward_meter_mean": 0.9769443273544312, "reward_meter_std": 0.023884175345301628, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.5175492167472839, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624143242835999, "reward_repeat_soft_std": 0.013508067466318607, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.22720351815223694, "reward_total_composite_mean": 0.6212908029556274, "reward_total_composite_std": 0.21466147899627686} {"timestamp_utc": "2026-04-13T10:32:06Z", "mode": "train", "global_step": 1283, "epoch": 0.12887995981918635, "loss": -0.0548, "grad_norm": 12.311016082763672, "learning_rate": 6.115151515151516e-06, "num_tokens": 2265598.0, "completions/mean_length": 47.75, "completions/min_length": 33.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.75, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.907586932182312, "rewards/meter/std": 0.226583793759346, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9689907431602478, "rewards/repeat_soft/std": 0.03646272420883179, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.6331464052200317, "rewards/total_composite/std": 0.13213784992694855, "reward": 0.6331464052200317, "reward_std": 0.13213784992694855, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18488241732120514, "sampling/sampling_logp_difference/max": 1.675678014755249, "sampling/importance_sampling_ratio/min": 0.18718121945858002, "sampling/importance_sampling_ratio/mean": 1.0197899341583252, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.362459436058998, "clip_ratio/low_mean": 0.162873312830925, "clip_ratio/low_min": 0.162873312830925, "clip_ratio/high_mean": 0.01844262331724167, "clip_ratio/high_max": 0.01844262331724167, "clip_ratio/region_mean": 0.18131593614816666, "reward_total_mean": 0.6331464052200317, "reward_meter_mean": 0.907586932182312, "reward_meter_std": 0.226583793759346, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9689907431602478, "reward_repeat_soft_std": 0.03646272420883179, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.6331464052200317, "reward_total_composite_std": 0.13213784992694855} {"timestamp_utc": "2026-04-13T10:32:13Z", "mode": "train", "global_step": 1284, "epoch": 0.12898041185334003, "loss": 0.0621, "grad_norm": 10.086344718933105, "learning_rate": 6.112121212121213e-06, "num_tokens": 2267807.0, "completions/mean_length": 90.125, "completions/min_length": 80.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.125, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.883400022983551, "rewards/meter/std": 0.2530997395515442, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9838672280311584, "rewards/repeat_soft/std": 0.01193454209715128, "rewards/judge_quality/mean": 0.6074999570846558, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.7078959941864014, "rewards/total_composite/std": 0.1462317556142807, "reward": 0.7078959941864014, "reward_std": 0.1462317407131195, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16071321070194244, "sampling/sampling_logp_difference/max": 2.1969804763793945, "sampling/importance_sampling_ratio/min": 0.1111382395029068, "sampling/importance_sampling_ratio/mean": 0.9974961280822754, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9173566922545433, "clip_ratio/low_mean": 0.05083981156349182, "clip_ratio/low_min": 0.05083981156349182, "clip_ratio/high_mean": 0.08948590233922005, "clip_ratio/high_max": 0.08948590233922005, "clip_ratio/region_mean": 0.14032571390271187, "reward_total_mean": 0.7078959941864014, "reward_meter_mean": 0.883400022983551, "reward_meter_std": 0.2530997395515442, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9838672280311584, "reward_repeat_soft_std": 0.01193454209715128, "reward_judge_quality_mean": 0.6074999570846558, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.7078959941864014, "reward_total_composite_std": 0.1462317556142807} {"timestamp_utc": "2026-04-13T10:32:20Z", "mode": "train", "global_step": 1285, "epoch": 0.12908086388749374, "loss": 0.0938, "grad_norm": 9.451906204223633, "learning_rate": 6.10909090909091e-06, "num_tokens": 2270093.0, "completions/mean_length": 86.75, "completions/min_length": 78.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.75, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.6015616059303284, "rewards/meter/std": 0.412428617477417, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9905730485916138, "rewards/repeat_soft/std": 0.007027474697679281, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.5160260200500488, "rewards/total_composite/std": 0.1082032173871994, "reward": 0.5160260200500488, "reward_std": 0.1082032099366188, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1459343284368515, "sampling/sampling_logp_difference/max": 2.7698051929473877, "sampling/importance_sampling_ratio/min": 0.06267420947551727, "sampling/importance_sampling_ratio/mean": 1.020920991897583, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8987893760204315, "clip_ratio/low_mean": 0.056004864163696766, "clip_ratio/low_min": 0.056004864163696766, "clip_ratio/high_mean": 0.06314123515039682, "clip_ratio/high_max": 0.06314123515039682, "clip_ratio/region_mean": 0.11914609931409359, "reward_total_mean": 0.5160260200500488, "reward_meter_mean": 0.6015616059303284, "reward_meter_std": 0.412428617477417, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9905730485916138, "reward_repeat_soft_std": 0.007027474697679281, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.5160260200500488, "reward_total_composite_std": 0.1082032173871994} {"timestamp_utc": "2026-04-13T10:32:26Z", "mode": "train", "global_step": 1286, "epoch": 0.12918131592164742, "loss": 0.0092, "grad_norm": 16.7330379486084, "learning_rate": 6.106060606060606e-06, "num_tokens": 2271688.0, "completions/mean_length": 41.375, "completions/min_length": 36.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.375, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.5061711072921753, "rewards/meter/std": 0.3633720874786377, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.991203784942627, "rewards/repeat_soft/std": 0.008159817196428776, "rewards/judge_quality/mean": 0.5824999809265137, "rewards/judge_quality/std": 0.23260943591594696, "rewards/total_composite/mean": 0.5223467350006104, "rewards/total_composite/std": 0.09643376618623734, "reward": 0.5223467350006104, "reward_std": 0.09643375873565674, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14730842411518097, "sampling/sampling_logp_difference/max": 2.860617160797119, "sampling/importance_sampling_ratio/min": 0.057233426719903946, "sampling/importance_sampling_ratio/mean": 1.0176209211349487, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.883740559220314, "clip_ratio/low_mean": 0.0753450570628047, "clip_ratio/low_min": 0.0753450570628047, "clip_ratio/high_mean": 0.08628917392343283, "clip_ratio/high_max": 0.08628917392343283, "clip_ratio/region_mean": 0.16163423098623753, "reward_total_mean": 0.5223467350006104, "reward_meter_mean": 0.5061711072921753, "reward_meter_std": 0.3633720874786377, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.991203784942627, "reward_repeat_soft_std": 0.008159817196428776, "reward_judge_quality_mean": 0.5824999809265137, "reward_judge_quality_std": 0.23260943591594696, "reward_total_composite_mean": 0.5223467350006104, "reward_total_composite_std": 0.09643376618623734} {"timestamp_utc": "2026-04-13T10:32:37Z", "mode": "train", "global_step": 1287, "epoch": 0.1292817679558011, "loss": -0.111, "grad_norm": 3.8122456073760986, "learning_rate": 6.103030303030304e-06, "num_tokens": 2273309.0, "completions/mean_length": 112.625, "completions/min_length": 49.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 55.57143020629883, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.7934470772743225, "rewards/meter/std": 0.28795894980430603, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9797966480255127, "rewards/repeat_soft/std": 0.02857290394604206, "rewards/judge_quality/mean": 0.7237499952316284, "rewards/judge_quality/std": 0.32487085461616516, "rewards/total_composite/mean": 0.6774677038192749, "rewards/total_composite/std": 0.3324925899505615, "reward": 0.6774677038192749, "reward_std": 0.3324925899505615, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16111302375793457, "sampling/sampling_logp_difference/max": 1.9014368057250977, "sampling/importance_sampling_ratio/min": 0.14935387670993805, "sampling/importance_sampling_ratio/mean": 0.9883249998092651, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5149192363023758, "clip_ratio/low_mean": 0.03316326625645161, "clip_ratio/low_min": 0.03316326625645161, "clip_ratio/high_mean": 0.09763261023908854, "clip_ratio/high_max": 0.09763261023908854, "clip_ratio/region_mean": 0.13079587649554014, "reward_total_mean": 0.6774677038192749, "reward_meter_mean": 0.7934470772743225, "reward_meter_std": 0.28795894980430603, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9797966480255127, "reward_repeat_soft_std": 0.02857290394604206, "reward_judge_quality_mean": 0.7237499952316284, "reward_judge_quality_std": 0.32487085461616516, "reward_total_composite_mean": 0.6774677038192749, "reward_total_composite_std": 0.3324925899505615} {"timestamp_utc": "2026-04-13T10:32:49Z", "mode": "train", "global_step": 1288, "epoch": 0.1293822199899548, "loss": -0.0769, "grad_norm": 2.889350652694702, "learning_rate": 6.1e-06, "num_tokens": 2274900.0, "completions/mean_length": 89.875, "completions/min_length": 25.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 29.571430206298828, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.6511855125427246, "rewards/meter/std": 0.36623069643974304, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9671874642372131, "rewards/repeat_soft/std": 0.013258260674774647, "rewards/judge_quality/mean": 0.4337500035762787, "rewards/judge_quality/std": 0.24663087725639343, "rewards/total_composite/mean": 0.5051541924476624, "rewards/total_composite/std": 0.24218609929084778, "reward": 0.5051541924476624, "reward_std": 0.24218609929084778, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1646328568458557, "sampling/sampling_logp_difference/max": 1.674455165863037, "sampling/importance_sampling_ratio/min": 0.18741025030612946, "sampling/importance_sampling_ratio/mean": 0.9861435294151306, "sampling/importance_sampling_ratio/max": 1.7221283912658691, "entropy": 1.0403777062892914, "clip_ratio/low_mean": 0.03987455181777477, "clip_ratio/low_min": 0.03987455181777477, "clip_ratio/high_mean": 0.1197044001892209, "clip_ratio/high_max": 0.1197044001892209, "clip_ratio/region_mean": 0.15957895200699568, "reward_total_mean": 0.5051541924476624, "reward_meter_mean": 0.6511855125427246, "reward_meter_std": 0.36623069643974304, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9671874642372131, "reward_repeat_soft_std": 0.013258260674774647, "reward_judge_quality_mean": 0.4337500035762787, "reward_judge_quality_std": 0.24663087725639343, "reward_total_composite_mean": 0.5051541924476624, "reward_total_composite_std": 0.24218609929084778} {"timestamp_utc": "2026-04-13T10:32:56Z", "mode": "train", "global_step": 1289, "epoch": 0.12948267202410849, "loss": 0.0438, "grad_norm": 13.759574890136719, "learning_rate": 6.096969696969698e-06, "num_tokens": 2276788.0, "completions/mean_length": 52.0, "completions/min_length": 48.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.0, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.247971773147583, "rewards/meter/std": 0.3090267777442932, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9975694417953491, "rewards/repeat_soft/std": 0.004525233991444111, "rewards/judge_quality/mean": 0.690000057220459, "rewards/judge_quality/std": 0.2343989461660385, "rewards/total_composite/mean": 0.44359105825424194, "rewards/total_composite/std": 0.10209126770496368, "reward": 0.44359105825424194, "reward_std": 0.10209126770496368, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13923315703868866, "sampling/sampling_logp_difference/max": 2.5892133712768555, "sampling/importance_sampling_ratio/min": 0.07507907599210739, "sampling/importance_sampling_ratio/mean": 0.9784464836120605, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5176977664232254, "clip_ratio/low_mean": 0.057573966681957245, "clip_ratio/low_min": 0.057573966681957245, "clip_ratio/high_mean": 0.056528182700276375, "clip_ratio/high_max": 0.056528182700276375, "clip_ratio/region_mean": 0.11410214938223362, "reward_total_mean": 0.44359105825424194, "reward_meter_mean": 0.247971773147583, "reward_meter_std": 0.3090267777442932, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9975694417953491, "reward_repeat_soft_std": 0.004525233991444111, "reward_judge_quality_mean": 0.690000057220459, "reward_judge_quality_std": 0.2343989461660385, "reward_total_composite_mean": 0.44359105825424194, "reward_total_composite_std": 0.10209126770496368} {"timestamp_utc": "2026-04-13T10:33:07Z", "mode": "train", "global_step": 1290, "epoch": 0.12958312405826217, "loss": -0.1377, "grad_norm": 2.128577470779419, "learning_rate": 6.0939393939393946e-06, "num_tokens": 2278437.0, "completions/mean_length": 174.125, "completions/min_length": 48.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 61.5, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.8028546571731567, "rewards/meter/std": 0.3164840340614319, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.2357022762298584, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9554349184036255, "rewards/repeat_soft/std": 0.03757408261299133, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.20694634318351746, "rewards/total_composite/mean": 0.47944241762161255, "rewards/total_composite/std": 0.30393680930137634, "reward": 0.47944241762161255, "reward_std": 0.30393680930137634, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14244621992111206, "sampling/sampling_logp_difference/max": 1.4288115501403809, "sampling/importance_sampling_ratio/min": 0.239593505859375, "sampling/importance_sampling_ratio/mean": 1.0086246728897095, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5359006002545357, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.11664778739213943, "clip_ratio/high_max": 0.11664778739213943, "clip_ratio/region_mean": 0.11664778739213943, "reward_total_mean": 0.47944241762161255, "reward_meter_mean": 0.8028546571731567, "reward_meter_std": 0.3164840340614319, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.2357022762298584, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9554349184036255, "reward_repeat_soft_std": 0.03757408261299133, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.20694634318351746, "reward_total_composite_mean": 0.47944241762161255, "reward_total_composite_std": 0.30393680930137634} {"timestamp_utc": "2026-04-13T10:33:13Z", "mode": "train", "global_step": 1291, "epoch": 0.12968357609241588, "loss": 0.0015, "grad_norm": 12.229418754577637, "learning_rate": 6.090909090909092e-06, "num_tokens": 2280580.0, "completions/mean_length": 74.875, "completions/min_length": 67.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.875, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.7657289505004883, "rewards/meter/std": 0.3279200792312622, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.920647382736206, "rewards/repeat_soft/std": 0.052993398159742355, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.517281174659729, "rewards/total_composite/std": 0.07782205939292908, "reward": 0.517281174659729, "reward_std": 0.07782205939292908, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14785484969615936, "sampling/sampling_logp_difference/max": 1.4468560218811035, "sampling/importance_sampling_ratio/min": 0.2353089302778244, "sampling/importance_sampling_ratio/mean": 0.9972316026687622, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9556903690099716, "clip_ratio/low_mean": 0.05760151706635952, "clip_ratio/low_min": 0.05760151706635952, "clip_ratio/high_mean": 0.0969934118911624, "clip_ratio/high_max": 0.0969934118911624, "clip_ratio/region_mean": 0.15459492895752192, "reward_total_mean": 0.517281174659729, "reward_meter_mean": 0.7657289505004883, "reward_meter_std": 0.3279200792312622, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.920647382736206, "reward_repeat_soft_std": 0.052993398159742355, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.517281174659729, "reward_total_composite_std": 0.07782205939292908} {"timestamp_utc": "2026-04-13T10:33:24Z", "mode": "train", "global_step": 1292, "epoch": 0.12978402812656956, "loss": -0.1627, "grad_norm": 3.7194430828094482, "learning_rate": 6.087878787878788e-06, "num_tokens": 2282395.0, "completions/mean_length": 124.875, "completions/min_length": 59.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 69.5714340209961, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.35180729627609253, "rewards/meter/std": 0.2963906526565552, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9926522374153137, "rewards/repeat_soft/std": 0.010722406208515167, "rewards/judge_quality/mean": 0.5987499952316284, "rewards/judge_quality/std": 0.3210001289844513, "rewards/total_composite/mean": 0.449861079454422, "rewards/total_composite/std": 0.22524167597293854, "reward": 0.449861079454422, "reward_std": 0.22524167597293854, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14167234301567078, "sampling/sampling_logp_difference/max": 2.0441176891326904, "sampling/importance_sampling_ratio/min": 0.12949439883232117, "sampling/importance_sampling_ratio/mean": 1.0059019327163696, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6685599088668823, "clip_ratio/low_mean": 0.04187853168696165, "clip_ratio/low_min": 0.04187853168696165, "clip_ratio/high_mean": 0.08393612876534462, "clip_ratio/high_max": 0.08393612876534462, "clip_ratio/region_mean": 0.12581466045230627, "reward_total_mean": 0.449861079454422, "reward_meter_mean": 0.35180729627609253, "reward_meter_std": 0.2963906526565552, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9926522374153137, "reward_repeat_soft_std": 0.010722406208515167, "reward_judge_quality_mean": 0.5987499952316284, "reward_judge_quality_std": 0.3210001289844513, "reward_total_composite_mean": 0.449861079454422, "reward_total_composite_std": 0.22524167597293854} {"timestamp_utc": "2026-04-13T10:33:30Z", "mode": "train", "global_step": 1293, "epoch": 0.12988448016072326, "loss": 0.0077, "grad_norm": 11.292625427246094, "learning_rate": 6.0848484848484855e-06, "num_tokens": 2284209.0, "completions/mean_length": 63.75, "completions/min_length": 56.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.75, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9202028512954712, "rewards/meter/std": 0.1847878098487854, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9812027812004089, "rewards/repeat_soft/std": 0.022314518690109253, "rewards/judge_quality/mean": 0.5325000286102295, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.6703628301620483, "rewards/total_composite/std": 0.12549921870231628, "reward": 0.6703628301620483, "reward_std": 0.12549921870231628, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15040865540504456, "sampling/sampling_logp_difference/max": 1.460923194885254, "sampling/importance_sampling_ratio/min": 0.2320219874382019, "sampling/importance_sampling_ratio/mean": 1.0040225982666016, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8783659860491753, "clip_ratio/low_mean": 0.08028423693031073, "clip_ratio/low_min": 0.08028423693031073, "clip_ratio/high_mean": 0.05312481801956892, "clip_ratio/high_max": 0.05312481801956892, "clip_ratio/region_mean": 0.13340905494987965, "reward_total_mean": 0.6703628301620483, "reward_meter_mean": 0.9202028512954712, "reward_meter_std": 0.1847878098487854, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9812027812004089, "reward_repeat_soft_std": 0.022314518690109253, "reward_judge_quality_mean": 0.5325000286102295, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.6703628301620483, "reward_total_composite_std": 0.12549921870231628} {"timestamp_utc": "2026-04-13T10:33:36Z", "mode": "train", "global_step": 1294, "epoch": 0.12998493219487695, "loss": 0.3254, "grad_norm": 20.19084358215332, "learning_rate": 6.081818181818182e-06, "num_tokens": 2285629.0, "completions/mean_length": 27.5, "completions/min_length": 20.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.5, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.684787929058075, "rewards/meter/std": 0.3552640378475189, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.953275203704834, "rewards/repeat_soft/std": 0.024769343435764313, "rewards/judge_quality/mean": 0.7362500429153442, "rewards/judge_quality/std": 0.25376805663108826, "rewards/total_composite/mean": 0.6545860767364502, "rewards/total_composite/std": 0.26531553268432617, "reward": 0.6545860767364502, "reward_std": 0.26531553268432617, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15634866058826447, "sampling/sampling_logp_difference/max": 1.5979595184326172, "sampling/importance_sampling_ratio/min": 0.20230890810489655, "sampling/importance_sampling_ratio/mean": 1.0324431657791138, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9607767537236214, "clip_ratio/low_mean": 0.10757675394415855, "clip_ratio/low_min": 0.10757675394415855, "clip_ratio/high_mean": 0.0832880437374115, "clip_ratio/high_max": 0.0832880437374115, "clip_ratio/region_mean": 0.19086479768157005, "reward_total_mean": 0.6545860767364502, "reward_meter_mean": 0.684787929058075, "reward_meter_std": 0.3552640378475189, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.953275203704834, "reward_repeat_soft_std": 0.024769343435764313, "reward_judge_quality_mean": 0.7362500429153442, "reward_judge_quality_std": 0.25376805663108826, "reward_total_composite_mean": 0.6545860767364502, "reward_total_composite_std": 0.26531553268432617} {"timestamp_utc": "2026-04-13T10:33:43Z", "mode": "train", "global_step": 1295, "epoch": 0.13008538422903063, "loss": 0.0639, "grad_norm": 8.543797492980957, "learning_rate": 6.07878787878788e-06, "num_tokens": 2287947.0, "completions/mean_length": 102.75, "completions/min_length": 85.0, "completions/max_length": 118.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.75, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 118.0, "rewards/meter/mean": 0.8280737996101379, "rewards/meter/std": 0.25559964776039124, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.979012668132782, "rewards/repeat_soft/std": 0.009831268340349197, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.5636317729949951, "rewards/total_composite/std": 0.10877915471792221, "reward": 0.5636317729949951, "reward_std": 0.10877914726734161, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15261870622634888, "sampling/sampling_logp_difference/max": 1.567774772644043, "sampling/importance_sampling_ratio/min": 0.20850864052772522, "sampling/importance_sampling_ratio/mean": 1.0170245170593262, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9373430162668228, "clip_ratio/low_mean": 0.03791691828519106, "clip_ratio/low_min": 0.03791691828519106, "clip_ratio/high_mean": 0.0900578573346138, "clip_ratio/high_max": 0.0900578573346138, "clip_ratio/region_mean": 0.12797477561980486, "reward_total_mean": 0.5636317729949951, "reward_meter_mean": 0.8280737996101379, "reward_meter_std": 0.25559964776039124, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.979012668132782, "reward_repeat_soft_std": 0.009831268340349197, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.5636317729949951, "reward_total_composite_std": 0.10877915471792221} {"timestamp_utc": "2026-04-13T10:33:49Z", "mode": "train", "global_step": 1296, "epoch": 0.13018583626318433, "loss": 0.0749, "grad_norm": 19.56597900390625, "learning_rate": 6.0757575757575755e-06, "num_tokens": 2289501.0, "completions/mean_length": 41.25, "completions/min_length": 34.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.25, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.8564885854721069, "rewards/meter/std": 0.26120027899742126, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9801363348960876, "rewards/repeat_soft/std": 0.010971256531774998, "rewards/judge_quality/mean": 0.48875001072883606, "rewards/judge_quality/std": 0.16591200232505798, "rewards/total_composite/mean": 0.624244213104248, "rewards/total_composite/std": 0.14069931209087372, "reward": 0.624244213104248, "reward_std": 0.14069931209087372, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18532513082027435, "sampling/sampling_logp_difference/max": 2.3976306915283203, "sampling/importance_sampling_ratio/min": 0.09093314409255981, "sampling/importance_sampling_ratio/mean": 1.0053044557571411, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0849749743938446, "clip_ratio/low_mean": 0.10320191085338593, "clip_ratio/low_min": 0.10320191085338593, "clip_ratio/high_mean": 0.07297545112669468, "clip_ratio/high_max": 0.07297545112669468, "clip_ratio/region_mean": 0.1761773619800806, "reward_total_mean": 0.624244213104248, "reward_meter_mean": 0.8564885854721069, "reward_meter_std": 0.26120027899742126, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9801363348960876, "reward_repeat_soft_std": 0.010971256531774998, "reward_judge_quality_mean": 0.48875001072883606, "reward_judge_quality_std": 0.16591200232505798, "reward_total_composite_mean": 0.624244213104248, "reward_total_composite_std": 0.14069931209087372} {"timestamp_utc": "2026-04-13T10:34:00Z", "mode": "train", "global_step": 1297, "epoch": 0.13028628829733802, "loss": -0.0931, "grad_norm": 2.960124969482422, "learning_rate": 6.072727272727274e-06, "num_tokens": 2290954.0, "completions/mean_length": 96.625, "completions/min_length": 29.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 37.28571701049805, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.5484809875488281, "rewards/meter/std": 0.34423375129699707, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9816007614135742, "rewards/repeat_soft/std": 0.02931581251323223, "rewards/judge_quality/mean": 0.5762499570846558, "rewards/judge_quality/std": 0.31513887643814087, "rewards/total_composite/mean": 0.48849669098854065, "rewards/total_composite/std": 0.21631859242916107, "reward": 0.48849669098854065, "reward_std": 0.21631857752799988, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1759599894285202, "sampling/sampling_logp_difference/max": 2.608419895172119, "sampling/importance_sampling_ratio/min": 0.07365082949399948, "sampling/importance_sampling_ratio/mean": 1.0088098049163818, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0217162147164345, "clip_ratio/low_mean": 0.01944444514811039, "clip_ratio/low_min": 0.01944444514811039, "clip_ratio/high_mean": 0.13144625257700682, "clip_ratio/high_max": 0.13144625257700682, "clip_ratio/region_mean": 0.1508906977251172, "reward_total_mean": 0.48849669098854065, "reward_meter_mean": 0.5484809875488281, "reward_meter_std": 0.34423375129699707, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9816007614135742, "reward_repeat_soft_std": 0.02931581251323223, "reward_judge_quality_mean": 0.5762499570846558, "reward_judge_quality_std": 0.31513887643814087, "reward_total_composite_mean": 0.48849669098854065, "reward_total_composite_std": 0.21631859242916107} {"timestamp_utc": "2026-04-13T10:34:12Z", "mode": "train", "global_step": 1298, "epoch": 0.13038674033149172, "loss": -0.1556, "grad_norm": 2.7274155616760254, "learning_rate": 6.06969696969697e-06, "num_tokens": 2292935.0, "completions/mean_length": 137.625, "completions/min_length": 72.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 84.14286041259766, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 100.0, "rewards/meter/mean": 0.7151336073875427, "rewards/meter/std": 0.3311758041381836, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.981122612953186, "rewards/repeat_soft/std": 0.020318450406193733, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.13593590259552002, "rewards/total_composite/mean": 0.47950494289398193, "rewards/total_composite/std": 0.20593856275081635, "reward": 0.47950494289398193, "reward_std": 0.20593856275081635, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16555652022361755, "sampling/sampling_logp_difference/max": 2.035820484161377, "sampling/importance_sampling_ratio/min": 0.1305733025074005, "sampling/importance_sampling_ratio/mean": 1.0415858030319214, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0952675566077232, "clip_ratio/low_mean": 0.025537634268403053, "clip_ratio/low_min": 0.025537634268403053, "clip_ratio/high_mean": 0.11700581107288599, "clip_ratio/high_max": 0.11700581107288599, "clip_ratio/region_mean": 0.14254344534128904, "reward_total_mean": 0.47950494289398193, "reward_meter_mean": 0.7151336073875427, "reward_meter_std": 0.3311758041381836, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.981122612953186, "reward_repeat_soft_std": 0.020318450406193733, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.13593590259552002, "reward_total_composite_mean": 0.47950494289398193, "reward_total_composite_std": 0.20593856275081635} {"timestamp_utc": "2026-04-13T10:34:18Z", "mode": "train", "global_step": 1299, "epoch": 0.1304871923656454, "loss": 0.056, "grad_norm": 13.681201934814453, "learning_rate": 6.066666666666667e-06, "num_tokens": 2294597.0, "completions/mean_length": 36.75, "completions/min_length": 31.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.75, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.33265426754951477, "rewards/meter/std": 0.28129392862319946, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9925545454025269, "rewards/repeat_soft/std": 0.01810423843562603, "rewards/judge_quality/mean": 0.53125, "rewards/judge_quality/std": 0.1865811049938202, "rewards/total_composite/mean": 0.45487838983535767, "rewards/total_composite/std": 0.07872791588306427, "reward": 0.45487838983535767, "reward_std": 0.07872791588306427, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14265796542167664, "sampling/sampling_logp_difference/max": 2.450737953186035, "sampling/importance_sampling_ratio/min": 0.08622992783784866, "sampling/importance_sampling_ratio/mean": 1.0143132209777832, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7138647735118866, "clip_ratio/low_mean": 0.060220772633329034, "clip_ratio/low_min": 0.060220772633329034, "clip_ratio/high_mean": 0.07355498988181353, "clip_ratio/high_max": 0.07355498988181353, "clip_ratio/region_mean": 0.13377576251514256, "reward_total_mean": 0.45487838983535767, "reward_meter_mean": 0.33265426754951477, "reward_meter_std": 0.28129392862319946, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9925545454025269, "reward_repeat_soft_std": 0.01810423843562603, "reward_judge_quality_mean": 0.53125, "reward_judge_quality_std": 0.1865811049938202, "reward_total_composite_mean": 0.45487838983535767, "reward_total_composite_std": 0.07872791588306427} {"timestamp_utc": "2026-04-13T10:34:29Z", "mode": "train", "global_step": 1300, "epoch": 0.13058764439979909, "loss": -0.1511, "grad_norm": 3.13342022895813, "learning_rate": 6.063636363636364e-06, "num_tokens": 2296608.0, "completions/mean_length": 124.375, "completions/min_length": 62.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 69.0, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.3995330035686493, "rewards/meter/std": 0.4067206382751465, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9935516119003296, "rewards/repeat_soft/std": 0.008096016943454742, "rewards/judge_quality/mean": 0.4449999928474426, "rewards/judge_quality/std": 0.21876277029514313, "rewards/total_composite/mean": 0.417496919631958, "rewards/total_composite/std": 0.19607169926166534, "reward": 0.417496919631958, "reward_std": 0.19607168436050415, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1340249627828598, "sampling/sampling_logp_difference/max": 1.2487568855285645, "sampling/importance_sampling_ratio/min": 0.2868611812591553, "sampling/importance_sampling_ratio/mean": 0.996375560760498, "sampling/importance_sampling_ratio/max": 1.8476996421813965, "entropy": 0.6958075687289238, "clip_ratio/low_mean": 0.04703459283336997, "clip_ratio/low_min": 0.04703459283336997, "clip_ratio/high_mean": 0.0712701603770256, "clip_ratio/high_max": 0.0712701603770256, "clip_ratio/region_mean": 0.11830475321039557, "reward_total_mean": 0.417496919631958, "reward_meter_mean": 0.3995330035686493, "reward_meter_std": 0.4067206382751465, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9935516119003296, "reward_repeat_soft_std": 0.008096016943454742, "reward_judge_quality_mean": 0.4449999928474426, "reward_judge_quality_std": 0.21876277029514313, "reward_total_composite_mean": 0.417496919631958, "reward_total_composite_std": 0.19607169926166534} {"timestamp_utc": "2026-04-13T10:35:21Z", "mode": "eval", "global_step": 1300, "epoch": 0.13058764439979909, "eval_loss": NaN, "eval_runtime": 52.7541, "eval_samples_per_second": 1.516, "eval_steps_per_second": 0.19, "eval_num_tokens": 2296608.0, "eval_completions/mean_length": 83.8875, "eval_completions/min_length": 27.0, "eval_completions/max_length": 224.1, "eval_completions/clipped_ratio": 0.05, "eval_completions/mean_terminated_length": 61.05892906188965, "eval_completions/min_terminated_length": 27.0, "eval_completions/max_terminated_length": 99.4, "eval_rewards/meter/mean": 0.6011181086301803, "eval_rewards/meter/std": 0.36246598362922666, "eval_rewards/count_adherence/mean": 0.9479166507720947, "eval_rewards/count_adherence/std": 0.08784548602998257, "eval_rewards/hard_gate/mean": 0.95, "eval_rewards/hard_gate/std": 0.11700168251991272, "eval_rewards/repeat_soft/mean": 0.951271939277649, "eval_rewards/repeat_soft/std": 0.05197468902915716, "eval_rewards/judge_quality/mean": 0.44524999260902404, "eval_rewards/judge_quality/std": 0.15283784195780753, "eval_rewards/total_composite/mean": 0.49333526790142057, "eval_rewards/total_composite/std": 0.17335330545902253, "eval_reward": 0.49333526790142057, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.07236771360039711, "eval_sampling/sampling_logp_difference/max": 1.009927797317505, "eval_sampling/importance_sampling_ratio/min": 0.38720138370990753, "eval_sampling/importance_sampling_ratio/mean": 1.0154764890670775, "eval_sampling/importance_sampling_ratio/max": 1.4079676747322083, "eval_entropy": 0.8205519914627075, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.49333526790142057, "eval_reward_meter_mean": 0.6011181086301803, "eval_reward_meter_std": 0.36246598362922666, "eval_reward_count_adherence_mean": 0.9479166507720947, "eval_reward_count_adherence_std": 0.08784548602998257, "eval_reward_hard_gate_mean": 0.95, "eval_reward_hard_gate_std": 0.11700168251991272, "eval_reward_repeat_soft_mean": 0.951271939277649, "eval_reward_repeat_soft_std": 0.05197468902915716, "eval_reward_judge_quality_mean": 0.44524999260902404, "eval_reward_judge_quality_std": 0.15283784195780753, "eval_reward_total_composite_mean": 0.49333526790142057, "eval_reward_total_composite_std": 0.17335330545902253} {"timestamp_utc": "2026-04-13T10:35:37Z", "mode": "train", "global_step": 1301, "epoch": 0.1306880964339528, "loss": -0.1863, "grad_norm": 2.081382989883423, "learning_rate": 6.060606060606061e-06, "num_tokens": 2298683.0, "completions/mean_length": 202.375, "completions/min_length": 90.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 99.16667175292969, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.4714367687702179, "rewards/meter/std": 0.26599934697151184, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.18898223340511322, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9643048048019409, "rewards/repeat_soft/std": 0.03838948532938957, "rewards/judge_quality/mean": 0.32749998569488525, "rewards/judge_quality/std": 0.17127670347690582, "rewards/total_composite/mean": 0.3530154228210449, "rewards/total_composite/std": 0.22604341804981232, "reward": 0.3530154228210449, "reward_std": 0.22604341804981232, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12762080132961273, "sampling/sampling_logp_difference/max": 1.3520245552062988, "sampling/importance_sampling_ratio/min": 0.258715957403183, "sampling/importance_sampling_ratio/mean": 1.0024654865264893, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7316770181059837, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.11028255522251129, "clip_ratio/high_max": 0.11028255522251129, "clip_ratio/region_mean": 0.11028255522251129, "reward_total_mean": 0.3530154228210449, "reward_meter_mean": 0.4714367687702179, "reward_meter_std": 0.26599934697151184, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.18898223340511322, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9643048048019409, "reward_repeat_soft_std": 0.03838948532938957, "reward_judge_quality_mean": 0.32749998569488525, "reward_judge_quality_std": 0.17127670347690582, "reward_total_composite_mean": 0.3530154228210449, "reward_total_composite_std": 0.22604341804981232} {"timestamp_utc": "2026-04-13T10:35:44Z", "mode": "train", "global_step": 1302, "epoch": 0.13078854846810647, "loss": 0.0757, "grad_norm": 13.769412994384766, "learning_rate": 6.057575757575757e-06, "num_tokens": 2300221.0, "completions/mean_length": 37.25, "completions/min_length": 28.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.25, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.6589182615280151, "rewards/meter/std": 0.40660297870635986, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9563091993331909, "rewards/repeat_soft/std": 0.03997744992375374, "rewards/judge_quality/mean": 0.5562499761581421, "rewards/judge_quality/std": 0.2249404937028885, "rewards/total_composite/mean": 0.5515223145484924, "rewards/total_composite/std": 0.11672757565975189, "reward": 0.5515223145484924, "reward_std": 0.11672757565975189, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11146819591522217, "sampling/sampling_logp_difference/max": 0.8875398635864258, "sampling/importance_sampling_ratio/min": 0.41166725754737854, "sampling/importance_sampling_ratio/mean": 1.0080918073654175, "sampling/importance_sampling_ratio/max": 1.6126405000686646, "entropy": 0.8147349208593369, "clip_ratio/low_mean": 0.020604395773261786, "clip_ratio/low_min": 0.020604395773261786, "clip_ratio/high_mean": 0.07930322573520243, "clip_ratio/high_max": 0.07930322573520243, "clip_ratio/region_mean": 0.09990762150846422, "reward_total_mean": 0.5515223145484924, "reward_meter_mean": 0.6589182615280151, "reward_meter_std": 0.40660297870635986, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9563091993331909, "reward_repeat_soft_std": 0.03997744992375374, "reward_judge_quality_mean": 0.5562499761581421, "reward_judge_quality_std": 0.2249404937028885, "reward_total_composite_mean": 0.5515223145484924, "reward_total_composite_std": 0.11672757565975189} {"timestamp_utc": "2026-04-13T10:35:56Z", "mode": "train", "global_step": 1303, "epoch": 0.13088900050226018, "loss": -0.1772, "grad_norm": 3.8782999515533447, "learning_rate": 6.0545454545454555e-06, "num_tokens": 2302401.0, "completions/mean_length": 146.5, "completions/min_length": 86.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 94.28572082519531, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.5795238018035889, "rewards/meter/std": 0.248986154794693, "rewards/count_adherence/mean": 0.8250000476837158, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9817588329315186, "rewards/repeat_soft/std": 0.017542414367198944, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.1524970829486847, "rewards/total_composite/mean": 0.42917606234550476, "rewards/total_composite/std": 0.1993083655834198, "reward": 0.42917606234550476, "reward_std": 0.1993083357810974, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1815846562385559, "sampling/sampling_logp_difference/max": 2.7994613647460938, "sampling/importance_sampling_ratio/min": 0.06084282696247101, "sampling/importance_sampling_ratio/mean": 1.001242756843567, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9739576280117035, "clip_ratio/low_mean": 0.06201563123613596, "clip_ratio/low_min": 0.06201563123613596, "clip_ratio/high_mean": 0.08822392486035824, "clip_ratio/high_max": 0.08822392486035824, "clip_ratio/region_mean": 0.1502395560964942, "reward_total_mean": 0.42917606234550476, "reward_meter_mean": 0.5795238018035889, "reward_meter_std": 0.248986154794693, "reward_count_adherence_mean": 0.8250000476837158, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9817588329315186, "reward_repeat_soft_std": 0.017542414367198944, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.1524970829486847, "reward_total_composite_mean": 0.42917606234550476, "reward_total_composite_std": 0.1993083655834198} {"timestamp_utc": "2026-04-13T10:36:03Z", "mode": "train", "global_step": 1304, "epoch": 0.13098945253641386, "loss": 0.0499, "grad_norm": 14.70734691619873, "learning_rate": 6.051515151515152e-06, "num_tokens": 2304378.0, "completions/mean_length": 66.125, "completions/min_length": 52.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.125, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.7551811933517456, "rewards/meter/std": 0.2874099314212799, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.13363061845302582, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9010825157165527, "rewards/repeat_soft/std": 0.05824727937579155, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.5380387306213379, "rewards/total_composite/std": 0.09783925861120224, "reward": 0.5380387306213379, "reward_std": 0.09783925861120224, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15492960810661316, "sampling/sampling_logp_difference/max": 2.2266108989715576, "sampling/importance_sampling_ratio/min": 0.10789347440004349, "sampling/importance_sampling_ratio/mean": 1.0085541009902954, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7587547898292542, "clip_ratio/low_mean": 0.07222104258835316, "clip_ratio/low_min": 0.07222104258835316, "clip_ratio/high_mean": 0.07400674372911453, "clip_ratio/high_max": 0.07400674372911453, "clip_ratio/region_mean": 0.1462277863174677, "reward_total_mean": 0.5380387306213379, "reward_meter_mean": 0.7551811933517456, "reward_meter_std": 0.2874099314212799, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.13363061845302582, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9010825157165527, "reward_repeat_soft_std": 0.05824727937579155, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.5380387306213379, "reward_total_composite_std": 0.09783925861120224} {"timestamp_utc": "2026-04-13T10:36:10Z", "mode": "train", "global_step": 1305, "epoch": 0.13108990457056754, "loss": 0.0479, "grad_norm": 8.494245529174805, "learning_rate": 6.048484848484849e-06, "num_tokens": 2306503.0, "completions/mean_length": 83.625, "completions/min_length": 78.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 83.625, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.8273429870605469, "rewards/meter/std": 0.21092940866947174, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.940427839756012, "rewards/repeat_soft/std": 0.06156879663467407, "rewards/judge_quality/mean": 0.5325000286102295, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.6176351308822632, "rewards/total_composite/std": 0.08360213041305542, "reward": 0.6176351308822632, "reward_std": 0.08360213786363602, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1396074891090393, "sampling/sampling_logp_difference/max": 2.808736801147461, "sampling/importance_sampling_ratio/min": 0.060281090438365936, "sampling/importance_sampling_ratio/mean": 1.0032211542129517, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7733359336853027, "clip_ratio/low_mean": 0.07828882802277803, "clip_ratio/low_min": 0.07828882802277803, "clip_ratio/high_mean": 0.05440238770097494, "clip_ratio/high_max": 0.05440238770097494, "clip_ratio/region_mean": 0.13269121572375298, "reward_total_mean": 0.6176351308822632, "reward_meter_mean": 0.8273429870605469, "reward_meter_std": 0.21092940866947174, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.940427839756012, "reward_repeat_soft_std": 0.06156879663467407, "reward_judge_quality_mean": 0.5325000286102295, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.6176351308822632, "reward_total_composite_std": 0.08360213041305542} {"timestamp_utc": "2026-04-13T10:36:17Z", "mode": "train", "global_step": 1306, "epoch": 0.13119035660472125, "loss": 0.0345, "grad_norm": 10.0457181930542, "learning_rate": 6.0454545454545456e-06, "num_tokens": 2308856.0, "completions/mean_length": 93.125, "completions/min_length": 76.0, "completions/max_length": 106.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.125, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.23517444729804993, "rewards/meter/std": 0.1635940819978714, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9481331706047058, "rewards/repeat_soft/std": 0.035619623959064484, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.39940470457077026, "rewards/total_composite/std": 0.051205825060606, "reward": 0.39940470457077026, "reward_std": 0.051205825060606, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1481638103723526, "sampling/sampling_logp_difference/max": 2.566396951675415, "sampling/importance_sampling_ratio/min": 0.07681180536746979, "sampling/importance_sampling_ratio/mean": 1.0056487321853638, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7814144939184189, "clip_ratio/low_mean": 0.05869795382022858, "clip_ratio/low_min": 0.05869795382022858, "clip_ratio/high_mean": 0.0744721395894885, "clip_ratio/high_max": 0.0744721395894885, "clip_ratio/region_mean": 0.13317009340971708, "reward_total_mean": 0.39940470457077026, "reward_meter_mean": 0.23517444729804993, "reward_meter_std": 0.1635940819978714, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9481331706047058, "reward_repeat_soft_std": 0.035619623959064484, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.39940470457077026, "reward_total_composite_std": 0.051205825060606} {"timestamp_utc": "2026-04-13T10:36:23Z", "mode": "train", "global_step": 1307, "epoch": 0.13129080863887493, "loss": 0.1057, "grad_norm": 18.926837921142578, "learning_rate": 6.042424242424243e-06, "num_tokens": 2310360.0, "completions/mean_length": 33.0, "completions/min_length": 29.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.0, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.613053560256958, "rewards/meter/std": 0.41506704688072205, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9923344850540161, "rewards/repeat_soft/std": 0.012143817730247974, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.6067043542861938, "rewards/total_composite/std": 0.20889125764369965, "reward": 0.6067043542861938, "reward_std": 0.20889124274253845, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12716905772686005, "sampling/sampling_logp_difference/max": 1.055168867111206, "sampling/importance_sampling_ratio/min": 0.45095694065093994, "sampling/importance_sampling_ratio/mean": 1.0153416395187378, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9925641939043999, "clip_ratio/low_mean": 0.046996163204312325, "clip_ratio/low_min": 0.046996163204312325, "clip_ratio/high_mean": 0.0647506876848638, "clip_ratio/high_max": 0.0647506876848638, "clip_ratio/region_mean": 0.11174685088917613, "reward_total_mean": 0.6067043542861938, "reward_meter_mean": 0.613053560256958, "reward_meter_std": 0.41506704688072205, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9923344850540161, "reward_repeat_soft_std": 0.012143817730247974, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.6067043542861938, "reward_total_composite_std": 0.20889125764369965} {"timestamp_utc": "2026-04-13T10:36:29Z", "mode": "train", "global_step": 1308, "epoch": 0.13139126067302864, "loss": 0.0313, "grad_norm": 11.330538749694824, "learning_rate": 6.039393939393939e-06, "num_tokens": 2312017.0, "completions/mean_length": 60.125, "completions/min_length": 53.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.125, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.46473997831344604, "rewards/meter/std": 0.3321572542190552, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9457338452339172, "rewards/repeat_soft/std": 0.03787406533956528, "rewards/judge_quality/mean": 0.5362499952316284, "rewards/judge_quality/std": 0.1524970978498459, "rewards/total_composite/mean": 0.4909104108810425, "rewards/total_composite/std": 0.0880972370505333, "reward": 0.4909104108810425, "reward_std": 0.0880972146987915, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15379510819911957, "sampling/sampling_logp_difference/max": 2.13840913772583, "sampling/importance_sampling_ratio/min": 0.1178421676158905, "sampling/importance_sampling_ratio/mean": 1.0030930042266846, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8838731646537781, "clip_ratio/low_mean": 0.08330471720546484, "clip_ratio/low_min": 0.08330471720546484, "clip_ratio/high_mean": 0.05535276886075735, "clip_ratio/high_max": 0.05535276886075735, "clip_ratio/region_mean": 0.1386574860662222, "reward_total_mean": 0.4909104108810425, "reward_meter_mean": 0.46473997831344604, "reward_meter_std": 0.3321572542190552, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9457338452339172, "reward_repeat_soft_std": 0.03787406533956528, "reward_judge_quality_mean": 0.5362499952316284, "reward_judge_quality_std": 0.1524970978498459, "reward_total_composite_mean": 0.4909104108810425, "reward_total_composite_std": 0.0880972370505333} {"timestamp_utc": "2026-04-13T10:36:36Z", "mode": "train", "global_step": 1309, "epoch": 0.13149171270718232, "loss": -0.0157, "grad_norm": 12.610530853271484, "learning_rate": 6.0363636363636365e-06, "num_tokens": 2313604.0, "completions/mean_length": 36.375, "completions/min_length": 31.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.375, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.5330278873443604, "rewards/meter/std": 0.3682415783405304, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.939886212348938, "rewards/repeat_soft/std": 0.05095674470067024, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.48956868052482605, "rewards/total_composite/std": 0.10176670551300049, "reward": 0.48956868052482605, "reward_std": 0.10176671296358109, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12218648195266724, "sampling/sampling_logp_difference/max": 1.1730799674987793, "sampling/importance_sampling_ratio/min": 0.30941247940063477, "sampling/importance_sampling_ratio/mean": 1.0113162994384766, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7353175655007362, "clip_ratio/low_mean": 0.08282269071787596, "clip_ratio/low_min": 0.08282269071787596, "clip_ratio/high_mean": 0.06039342097938061, "clip_ratio/high_max": 0.06039342097938061, "clip_ratio/region_mean": 0.14321611169725657, "reward_total_mean": 0.48956868052482605, "reward_meter_mean": 0.5330278873443604, "reward_meter_std": 0.3682415783405304, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.939886212348938, "reward_repeat_soft_std": 0.05095674470067024, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.48956868052482605, "reward_total_composite_std": 0.10176670551300049} {"timestamp_utc": "2026-04-13T10:36:47Z", "mode": "train", "global_step": 1310, "epoch": 0.131592164741336, "loss": -0.1008, "grad_norm": 4.485108852386475, "learning_rate": 6.033333333333335e-06, "num_tokens": 2315242.0, "completions/mean_length": 101.75, "completions/min_length": 35.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 43.142860412597656, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.18399572372436523, "rewards/meter/std": 0.15946964919567108, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9867174625396729, "rewards/repeat_soft/std": 0.011840468272566795, "rewards/judge_quality/mean": 0.48124998807907104, "rewards/judge_quality/std": 0.2531762421131134, "rewards/total_composite/mean": 0.3581715226173401, "rewards/total_composite/std": 0.17232245206832886, "reward": 0.3581715226173401, "reward_std": 0.17232243716716766, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1770992875099182, "sampling/sampling_logp_difference/max": 1.5108189582824707, "sampling/importance_sampling_ratio/min": 0.22072914242744446, "sampling/importance_sampling_ratio/mean": 1.0185973644256592, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7373760268092155, "clip_ratio/low_mean": 0.057486264035105705, "clip_ratio/low_min": 0.057486264035105705, "clip_ratio/high_mean": 0.08373156748712063, "clip_ratio/high_max": 0.08373156748712063, "clip_ratio/region_mean": 0.14121783152222633, "reward_total_mean": 0.3581715226173401, "reward_meter_mean": 0.18399572372436523, "reward_meter_std": 0.15946964919567108, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9867174625396729, "reward_repeat_soft_std": 0.011840468272566795, "reward_judge_quality_mean": 0.48124998807907104, "reward_judge_quality_std": 0.2531762421131134, "reward_total_composite_mean": 0.3581715226173401, "reward_total_composite_std": 0.17232245206832886} {"timestamp_utc": "2026-04-13T10:36:59Z", "mode": "train", "global_step": 1311, "epoch": 0.1316926167754897, "loss": -0.0929, "grad_norm": 2.16886830329895, "learning_rate": 6.030303030303031e-06, "num_tokens": 2317169.0, "completions/mean_length": 226.875, "completions/min_length": 52.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 55.79999923706055, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.5467915534973145, "rewards/meter/std": 0.3904469907283783, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.5175492167472839, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6797391176223755, "rewards/repeat_soft/std": 0.3173426687717438, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.2834985554218292, "rewards/total_composite/mean": 0.4485192596912384, "rewards/total_composite/std": 0.33092156052589417, "reward": 0.4485192596912384, "reward_std": 0.33092156052589417, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09218810498714447, "sampling/sampling_logp_difference/max": 1.2387669086456299, "sampling/importance_sampling_ratio/min": 0.28974127769470215, "sampling/importance_sampling_ratio/mean": 0.9867146611213684, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.22532642632722855, "clip_ratio/low_mean": 0.021189458668231964, "clip_ratio/low_min": 0.021189458668231964, "clip_ratio/high_mean": 0.02835648157633841, "clip_ratio/high_max": 0.02835648157633841, "clip_ratio/region_mean": 0.049545940244570374, "reward_total_mean": 0.4485192596912384, "reward_meter_mean": 0.5467915534973145, "reward_meter_std": 0.3904469907283783, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.5175492167472839, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6797391176223755, "reward_repeat_soft_std": 0.3173426687717438, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.2834985554218292, "reward_total_composite_mean": 0.4485192596912384, "reward_total_composite_std": 0.33092156052589417} {"timestamp_utc": "2026-04-13T10:37:05Z", "mode": "train", "global_step": 1312, "epoch": 0.1317930688096434, "loss": 0.03, "grad_norm": 11.899909973144531, "learning_rate": 6.027272727272728e-06, "num_tokens": 2318847.0, "completions/mean_length": 37.75, "completions/min_length": 33.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.75, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.7507766485214233, "rewards/meter/std": 0.22238843142986298, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9750610589981079, "rewards/repeat_soft/std": 0.03612116351723671, "rewards/judge_quality/mean": 0.48624998331069946, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.5757731199264526, "rewards/total_composite/std": 0.07389680296182632, "reward": 0.5757731199264526, "reward_std": 0.07389680296182632, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1200757697224617, "sampling/sampling_logp_difference/max": 2.22994327545166, "sampling/importance_sampling_ratio/min": 0.10753452777862549, "sampling/importance_sampling_ratio/mean": 0.9822611808776855, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5109836533665657, "clip_ratio/low_mean": 0.03545725205913186, "clip_ratio/low_min": 0.03545725205913186, "clip_ratio/high_mean": 0.06825267663225532, "clip_ratio/high_max": 0.06825267663225532, "clip_ratio/region_mean": 0.10370992869138718, "reward_total_mean": 0.5757731199264526, "reward_meter_mean": 0.7507766485214233, "reward_meter_std": 0.22238843142986298, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9750610589981079, "reward_repeat_soft_std": 0.03612116351723671, "reward_judge_quality_mean": 0.48624998331069946, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.5757731199264526, "reward_total_composite_std": 0.07389680296182632} {"timestamp_utc": "2026-04-13T10:37:12Z", "mode": "train", "global_step": 1313, "epoch": 0.13189352084379707, "loss": 0.0006, "grad_norm": 13.080678939819336, "learning_rate": 6.024242424242425e-06, "num_tokens": 2320426.0, "completions/mean_length": 25.375, "completions/min_length": 24.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.375, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.7285583019256592, "rewards/meter/std": 0.31013527512550354, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9292786121368408, "rewards/repeat_soft/std": 0.05441930517554283, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.5627000331878662, "rewards/total_composite/std": 0.09881270676851273, "reward": 0.5627000331878662, "reward_std": 0.09881272166967392, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1227404922246933, "sampling/sampling_logp_difference/max": 1.3473396301269531, "sampling/importance_sampling_ratio/min": 0.25993087887763977, "sampling/importance_sampling_ratio/mean": 0.9929416179656982, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5725177004933357, "clip_ratio/low_mean": 0.03541666641831398, "clip_ratio/low_min": 0.03541666641831398, "clip_ratio/high_mean": 0.10167353507131338, "clip_ratio/high_max": 0.10167353507131338, "clip_ratio/region_mean": 0.13709020148962736, "reward_total_mean": 0.5627000331878662, "reward_meter_mean": 0.7285583019256592, "reward_meter_std": 0.31013527512550354, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9292786121368408, "reward_repeat_soft_std": 0.05441930517554283, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.5627000331878662, "reward_total_composite_std": 0.09881270676851273} {"timestamp_utc": "2026-04-13T10:37:18Z", "mode": "train", "global_step": 1314, "epoch": 0.13199397287795078, "loss": 0.0141, "grad_norm": 18.13827896118164, "learning_rate": 6.021212121212122e-06, "num_tokens": 2322019.0, "completions/mean_length": 42.125, "completions/min_length": 36.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.125, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.14196985960006714, "rewards/meter/std": 0.1567562371492386, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9805315732955933, "rewards/repeat_soft/std": 0.02711394801735878, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.1940544992685318, "rewards/total_composite/mean": 0.3856370151042938, "rewards/total_composite/std": 0.04208798334002495, "reward": 0.3856370151042938, "reward_std": 0.04208798334002495, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17558568716049194, "sampling/sampling_logp_difference/max": 1.8940314054489136, "sampling/importance_sampling_ratio/min": 0.1504640132188797, "sampling/importance_sampling_ratio/mean": 1.0315899848937988, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7719564884901047, "clip_ratio/low_mean": 0.1019074134528637, "clip_ratio/low_min": 0.1019074134528637, "clip_ratio/high_mean": 0.03083881549537182, "clip_ratio/high_max": 0.03083881549537182, "clip_ratio/region_mean": 0.1327462289482355, "reward_total_mean": 0.3856370151042938, "reward_meter_mean": 0.14196985960006714, "reward_meter_std": 0.1567562371492386, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9805315732955933, "reward_repeat_soft_std": 0.02711394801735878, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.1940544992685318, "reward_total_composite_mean": 0.3856370151042938, "reward_total_composite_std": 0.04208798334002495} {"timestamp_utc": "2026-04-13T10:37:24Z", "mode": "train", "global_step": 1315, "epoch": 0.13209442491210446, "loss": 0.0066, "grad_norm": 21.144624710083008, "learning_rate": 6.018181818181818e-06, "num_tokens": 2323419.0, "completions/mean_length": 27.0, "completions/min_length": 24.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.0, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.3701016902923584, "rewards/meter/std": 0.3760126233100891, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9868196249008179, "rewards/repeat_soft/std": 0.011254421435296535, "rewards/judge_quality/mean": 0.7312500476837158, "rewards/judge_quality/std": 0.20760111510753632, "rewards/total_composite/mean": 0.5289339423179626, "rewards/total_composite/std": 0.20768073201179504, "reward": 0.5289339423179626, "reward_std": 0.20768074691295624, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1722986400127411, "sampling/sampling_logp_difference/max": 1.8542876243591309, "sampling/importance_sampling_ratio/min": 0.15656444430351257, "sampling/importance_sampling_ratio/mean": 1.019766926765442, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9304405450820923, "clip_ratio/low_mean": 0.1068057743832469, "clip_ratio/low_min": 0.1068057743832469, "clip_ratio/high_mean": 0.03017241321504116, "clip_ratio/high_max": 0.03017241321504116, "clip_ratio/region_mean": 0.13697818759828806, "reward_total_mean": 0.5289339423179626, "reward_meter_mean": 0.3701016902923584, "reward_meter_std": 0.3760126233100891, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9868196249008179, "reward_repeat_soft_std": 0.011254421435296535, "reward_judge_quality_mean": 0.7312500476837158, "reward_judge_quality_std": 0.20760111510753632, "reward_total_composite_mean": 0.5289339423179626, "reward_total_composite_std": 0.20768073201179504} {"timestamp_utc": "2026-04-13T10:37:35Z", "mode": "train", "global_step": 1316, "epoch": 0.13219487694625817, "loss": -0.1241, "grad_norm": 4.055689811706543, "learning_rate": 6.015151515151516e-06, "num_tokens": 2325197.0, "completions/mean_length": 119.25, "completions/min_length": 58.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 63.142860412597656, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.563539981842041, "rewards/meter/std": 0.3978107273578644, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.930457353591919, "rewards/repeat_soft/std": 0.04242706298828125, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.44667279720306396, "rewards/total_composite/std": 0.20369835197925568, "reward": 0.44667279720306396, "reward_std": 0.20369835197925568, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16629445552825928, "sampling/sampling_logp_difference/max": 1.4497017860412598, "sampling/importance_sampling_ratio/min": 0.23464025557041168, "sampling/importance_sampling_ratio/mean": 1.0269790887832642, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.208445429801941, "clip_ratio/low_mean": 0.03451771289110184, "clip_ratio/low_min": 0.03451771289110184, "clip_ratio/high_mean": 0.08997233491390944, "clip_ratio/high_max": 0.08997233491390944, "clip_ratio/region_mean": 0.12449004780501127, "reward_total_mean": 0.44667279720306396, "reward_meter_mean": 0.563539981842041, "reward_meter_std": 0.3978107273578644, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.930457353591919, "reward_repeat_soft_std": 0.04242706298828125, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.44667279720306396, "reward_total_composite_std": 0.20369835197925568} {"timestamp_utc": "2026-04-13T10:37:49Z", "mode": "train", "global_step": 1317, "epoch": 0.13229532898041185, "loss": -0.1629, "grad_norm": 2.7149338722229004, "learning_rate": 6.012121212121213e-06, "num_tokens": 2327099.0, "completions/mean_length": 128.75, "completions/min_length": 64.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 74.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.32614681124687195, "rewards/meter/std": 0.27816957235336304, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.2651650309562683, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9868057370185852, "rewards/repeat_soft/std": 0.01320960745215416, "rewards/judge_quality/mean": 0.5362499952316284, "rewards/judge_quality/std": 0.29731839895248413, "rewards/total_composite/mean": 0.4235251843929291, "rewards/total_composite/std": 0.1879330724477768, "reward": 0.4235251843929291, "reward_std": 0.1879330724477768, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12032223492860794, "sampling/sampling_logp_difference/max": 2.3049678802490234, "sampling/importance_sampling_ratio/min": 0.09976200759410858, "sampling/importance_sampling_ratio/mean": 1.0027388334274292, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5216909050941467, "clip_ratio/low_mean": 0.020090090576559305, "clip_ratio/low_min": 0.020090090576559305, "clip_ratio/high_mean": 0.06276623345911503, "clip_ratio/high_max": 0.06276623345911503, "clip_ratio/region_mean": 0.08285632403567433, "reward_total_mean": 0.4235251843929291, "reward_meter_mean": 0.32614681124687195, "reward_meter_std": 0.27816957235336304, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.2651650309562683, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9868057370185852, "reward_repeat_soft_std": 0.01320960745215416, "reward_judge_quality_mean": 0.5362499952316284, "reward_judge_quality_std": 0.29731839895248413, "reward_total_composite_mean": 0.4235251843929291, "reward_total_composite_std": 0.1879330724477768} {"timestamp_utc": "2026-04-13T10:37:55Z", "mode": "train", "global_step": 1318, "epoch": 0.13239578101456553, "loss": -0.0521, "grad_norm": 12.158468246459961, "learning_rate": 6.00909090909091e-06, "num_tokens": 2328686.0, "completions/mean_length": 54.375, "completions/min_length": 36.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.375, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.4651309549808502, "rewards/meter/std": 0.2966498136520386, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9423577785491943, "rewards/repeat_soft/std": 0.05423344671726227, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.49903154373168945, "rewards/total_composite/std": 0.13276229798793793, "reward": 0.49903154373168945, "reward_std": 0.13276229798793793, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12698955833911896, "sampling/sampling_logp_difference/max": 1.449594497680664, "sampling/importance_sampling_ratio/min": 0.23466543853282928, "sampling/importance_sampling_ratio/mean": 0.9886088967323303, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6243526786565781, "clip_ratio/low_mean": 0.043492653872817755, "clip_ratio/low_min": 0.043492653872817755, "clip_ratio/high_mean": 0.08115777466446161, "clip_ratio/high_max": 0.08115777466446161, "clip_ratio/region_mean": 0.12465042853727937, "reward_total_mean": 0.49903154373168945, "reward_meter_mean": 0.4651309549808502, "reward_meter_std": 0.2966498136520386, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9423577785491943, "reward_repeat_soft_std": 0.05423344671726227, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.49903154373168945, "reward_total_composite_std": 0.13276229798793793} {"timestamp_utc": "2026-04-13T10:38:02Z", "mode": "train", "global_step": 1319, "epoch": 0.13249623304871924, "loss": -0.0504, "grad_norm": 11.741623878479004, "learning_rate": 6.0060606060606065e-06, "num_tokens": 2330593.0, "completions/mean_length": 64.375, "completions/min_length": 57.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.375, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.11510743200778961, "rewards/meter/std": 0.09939873963594437, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9502577185630798, "rewards/repeat_soft/std": 0.05569569766521454, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.37396299839019775, "rewards/total_composite/std": 0.02319924347102642, "reward": 0.37396299839019775, "reward_std": 0.02319924719631672, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17710459232330322, "sampling/sampling_logp_difference/max": 4.657275676727295, "sampling/importance_sampling_ratio/min": 0.009492287412285805, "sampling/importance_sampling_ratio/mean": 0.989753246307373, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.77280592918396, "clip_ratio/low_mean": 0.06375110056251287, "clip_ratio/low_min": 0.06375110056251287, "clip_ratio/high_mean": 0.05356408003717661, "clip_ratio/high_max": 0.05356408003717661, "clip_ratio/region_mean": 0.11731518059968948, "reward_total_mean": 0.37396299839019775, "reward_meter_mean": 0.11510743200778961, "reward_meter_std": 0.09939873963594437, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9502577185630798, "reward_repeat_soft_std": 0.05569569766521454, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.37396299839019775, "reward_total_composite_std": 0.02319924347102642} {"timestamp_utc": "2026-04-13T10:38:09Z", "mode": "train", "global_step": 1320, "epoch": 0.13259668508287292, "loss": 0.5899, "grad_norm": 10.33740234375, "learning_rate": 6.003030303030304e-06, "num_tokens": 2332245.0, "completions/mean_length": 59.5, "completions/min_length": 24.0, "completions/max_length": 183.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.5, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 183.0, "rewards/meter/mean": 0.29521444439888, "rewards/meter/std": 0.3094203472137451, "rewards/count_adherence/mean": 0.6875, "rewards/count_adherence/std": 0.45806270837783813, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9167537093162537, "rewards/repeat_soft/std": 0.11579932272434235, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.1940544992685318, "rewards/total_composite/mean": 0.3447820842266083, "rewards/total_composite/std": 0.11164502054452896, "reward": 0.3447820842266083, "reward_std": 0.11164502054452896, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07848409563302994, "sampling/sampling_logp_difference/max": 1.6388883590698242, "sampling/importance_sampling_ratio/min": 0.19419580698013306, "sampling/importance_sampling_ratio/mean": 1.0154354572296143, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.468915194272995, "clip_ratio/low_mean": 0.0402972879819572, "clip_ratio/low_min": 0.0402972879819572, "clip_ratio/high_mean": 0.050527742598205805, "clip_ratio/high_max": 0.050527742598205805, "clip_ratio/region_mean": 0.090825030580163, "reward_total_mean": 0.3447820842266083, "reward_meter_mean": 0.29521444439888, "reward_meter_std": 0.3094203472137451, "reward_count_adherence_mean": 0.6875, "reward_count_adherence_std": 0.45806270837783813, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9167537093162537, "reward_repeat_soft_std": 0.11579932272434235, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.1940544992685318, "reward_total_composite_mean": 0.3447820842266083, "reward_total_composite_std": 0.11164502054452896} {"timestamp_utc": "2026-04-13T10:38:16Z", "mode": "train", "global_step": 1321, "epoch": 0.13269713711702663, "loss": 0.0895, "grad_norm": 21.839662551879883, "learning_rate": 6e-06, "num_tokens": 2333782.0, "completions/mean_length": 16.125, "completions/min_length": 12.0, "completions/max_length": 23.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 16.125, "completions/min_terminated_length": 12.0, "completions/max_terminated_length": 23.0, "rewards/meter/mean": 0.6248440742492676, "rewards/meter/std": 0.3955559730529785, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.541622519493103, "rewards/total_composite/std": 0.1336933970451355, "reward": 0.541622519493103, "reward_std": 0.1336933970451355, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1574832648038864, "sampling/sampling_logp_difference/max": 1.6570532321929932, "sampling/importance_sampling_ratio/min": 0.19070009887218475, "sampling/importance_sampling_ratio/mean": 0.9786161780357361, "sampling/importance_sampling_ratio/max": 1.7580597400665283, "entropy": 0.9691708460450172, "clip_ratio/low_mean": 0.05115089565515518, "clip_ratio/low_min": 0.05115089565515518, "clip_ratio/high_mean": 0.13058911636471748, "clip_ratio/high_max": 0.13058911636471748, "clip_ratio/region_mean": 0.18174001201987267, "reward_total_mean": 0.541622519493103, "reward_meter_mean": 0.6248440742492676, "reward_meter_std": 0.3955559730529785, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.541622519493103, "reward_total_composite_std": 0.1336933970451355} {"timestamp_utc": "2026-04-13T10:38:23Z", "mode": "train", "global_step": 1322, "epoch": 0.1327975891511803, "loss": -0.0002, "grad_norm": 9.119623184204102, "learning_rate": 5.996969696969697e-06, "num_tokens": 2336046.0, "completions/mean_length": 87.0, "completions/min_length": 78.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.0, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.2986004650592804, "rewards/meter/std": 0.33540481328964233, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9160292148590088, "rewards/repeat_soft/std": 0.022331031039357185, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.22720350325107574, "rewards/total_composite/mean": 0.4448355734348297, "rewards/total_composite/std": 0.13325370848178864, "reward": 0.4448355734348297, "reward_std": 0.13325372338294983, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16736315190792084, "sampling/sampling_logp_difference/max": 1.9393364191055298, "sampling/importance_sampling_ratio/min": 0.14379934966564178, "sampling/importance_sampling_ratio/mean": 0.9967644214630127, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7418034449219704, "clip_ratio/low_mean": 0.08746557682752609, "clip_ratio/low_min": 0.08746557682752609, "clip_ratio/high_mean": 0.04391687922179699, "clip_ratio/high_max": 0.04391687922179699, "clip_ratio/region_mean": 0.13138245604932308, "reward_total_mean": 0.4448355734348297, "reward_meter_mean": 0.2986004650592804, "reward_meter_std": 0.33540481328964233, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9160292148590088, "reward_repeat_soft_std": 0.022331031039357185, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.22720350325107574, "reward_total_composite_mean": 0.4448355734348297, "reward_total_composite_std": 0.13325370848178864} {"timestamp_utc": "2026-04-13T10:38:29Z", "mode": "train", "global_step": 1323, "epoch": 0.132898041185334, "loss": -0.0499, "grad_norm": 14.36684799194336, "learning_rate": 5.993939393939394e-06, "num_tokens": 2337489.0, "completions/mean_length": 37.375, "completions/min_length": 33.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.375, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.7412725687026978, "rewards/meter/std": 0.3410572409629822, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9899576902389526, "rewards/repeat_soft/std": 0.009993739426136017, "rewards/judge_quality/mean": 0.5062500238418579, "rewards/judge_quality/std": 0.15080145001411438, "rewards/total_composite/mean": 0.5912611484527588, "rewards/total_composite/std": 0.12845481932163239, "reward": 0.5912611484527588, "reward_std": 0.12845481932163239, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14535550773143768, "sampling/sampling_logp_difference/max": 1.510312557220459, "sampling/importance_sampling_ratio/min": 0.2208409458398819, "sampling/importance_sampling_ratio/mean": 1.018187403678894, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8165575861930847, "clip_ratio/low_mean": 0.035386763513088226, "clip_ratio/low_min": 0.035386763513088226, "clip_ratio/high_mean": 0.08853822201490402, "clip_ratio/high_max": 0.08853822201490402, "clip_ratio/region_mean": 0.12392498552799225, "reward_total_mean": 0.5912611484527588, "reward_meter_mean": 0.7412725687026978, "reward_meter_std": 0.3410572409629822, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9899576902389526, "reward_repeat_soft_std": 0.009993739426136017, "reward_judge_quality_mean": 0.5062500238418579, "reward_judge_quality_std": 0.15080145001411438, "reward_total_composite_mean": 0.5912611484527588, "reward_total_composite_std": 0.12845481932163239} {"timestamp_utc": "2026-04-13T10:38:36Z", "mode": "train", "global_step": 1324, "epoch": 0.1329984932194877, "loss": 0.0343, "grad_norm": 11.929206848144531, "learning_rate": 5.990909090909092e-06, "num_tokens": 2339127.0, "completions/mean_length": 37.75, "completions/min_length": 34.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.75, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.7784693241119385, "rewards/meter/std": 0.3361956775188446, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9474538564682007, "rewards/repeat_soft/std": 0.060473356395959854, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.6314690709114075, "rewards/total_composite/std": 0.19294258952140808, "reward": 0.6314690709114075, "reward_std": 0.1929425746202469, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11215405911207199, "sampling/sampling_logp_difference/max": 1.9784116744995117, "sampling/importance_sampling_ratio/min": 0.1382887065410614, "sampling/importance_sampling_ratio/mean": 1.0129324197769165, "sampling/importance_sampling_ratio/max": 1.9263604879379272, "entropy": 0.7627578303217888, "clip_ratio/low_mean": 0.07707995735108852, "clip_ratio/low_min": 0.07707995735108852, "clip_ratio/high_mean": 0.040570175275206566, "clip_ratio/high_max": 0.040570175275206566, "clip_ratio/region_mean": 0.11765013262629509, "reward_total_mean": 0.6314690709114075, "reward_meter_mean": 0.7784693241119385, "reward_meter_std": 0.3361956775188446, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9474538564682007, "reward_repeat_soft_std": 0.060473356395959854, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.6314690709114075, "reward_total_composite_std": 0.19294258952140808} {"timestamp_utc": "2026-04-13T10:38:43Z", "mode": "train", "global_step": 1325, "epoch": 0.13309894525364138, "loss": 0.0459, "grad_norm": 11.517350196838379, "learning_rate": 5.987878787878788e-06, "num_tokens": 2341170.0, "completions/mean_length": 75.375, "completions/min_length": 64.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.375, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.3564927875995636, "rewards/meter/std": 0.31763216853141785, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.895226240158081, "rewards/repeat_soft/std": 0.06435451656579971, "rewards/judge_quality/mean": 0.5437500476837158, "rewards/judge_quality/std": 0.1765492856502533, "rewards/total_composite/mean": 0.41336947679519653, "rewards/total_composite/std": 0.22498710453510284, "reward": 0.41336947679519653, "reward_std": 0.22498708963394165, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13874313235282898, "sampling/sampling_logp_difference/max": 2.033254861831665, "sampling/importance_sampling_ratio/min": 0.13090874254703522, "sampling/importance_sampling_ratio/mean": 1.014022707939148, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7146521620452404, "clip_ratio/low_mean": 0.08022316731512547, "clip_ratio/low_min": 0.08022316731512547, "clip_ratio/high_mean": 0.05571772065013647, "clip_ratio/high_max": 0.05571772065013647, "clip_ratio/region_mean": 0.13594088796526194, "reward_total_mean": 0.41336947679519653, "reward_meter_mean": 0.3564927875995636, "reward_meter_std": 0.31763216853141785, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.895226240158081, "reward_repeat_soft_std": 0.06435451656579971, "reward_judge_quality_mean": 0.5437500476837158, "reward_judge_quality_std": 0.1765492856502533, "reward_total_composite_mean": 0.41336947679519653, "reward_total_composite_std": 0.22498710453510284} {"timestamp_utc": "2026-04-13T10:38:49Z", "mode": "train", "global_step": 1326, "epoch": 0.1331993972877951, "loss": 0.0946, "grad_norm": 14.53357219696045, "learning_rate": 5.984848484848486e-06, "num_tokens": 2343063.0, "completions/mean_length": 43.625, "completions/min_length": 38.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.625, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.567968487739563, "rewards/meter/std": 0.41750046610832214, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9909164905548096, "rewards/repeat_soft/std": 0.011348268948495388, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.5037947297096252, "rewards/total_composite/std": 0.11363547295331955, "reward": 0.5037947297096252, "reward_std": 0.11363547295331955, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14599864184856415, "sampling/sampling_logp_difference/max": 2.054823637008667, "sampling/importance_sampling_ratio/min": 0.1281154304742813, "sampling/importance_sampling_ratio/mean": 0.9894979000091553, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8291291519999504, "clip_ratio/low_mean": 0.046416791155934334, "clip_ratio/low_min": 0.046416791155934334, "clip_ratio/high_mean": 0.09215929359197617, "clip_ratio/high_max": 0.09215929359197617, "clip_ratio/region_mean": 0.1385760847479105, "reward_total_mean": 0.5037947297096252, "reward_meter_mean": 0.567968487739563, "reward_meter_std": 0.41750046610832214, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9909164905548096, "reward_repeat_soft_std": 0.011348268948495388, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.5037947297096252, "reward_total_composite_std": 0.11363547295331955} {"timestamp_utc": "2026-04-13T10:38:55Z", "mode": "train", "global_step": 1327, "epoch": 0.13329984932194877, "loss": -0.0002, "grad_norm": 16.082365036010742, "learning_rate": 5.981818181818182e-06, "num_tokens": 2344393.0, "completions/mean_length": 18.25, "completions/min_length": 17.0, "completions/max_length": 21.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 18.25, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 21.0, "rewards/meter/mean": 0.9682329893112183, "rewards/meter/std": 0.044429961591959, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.7628706693649292, "rewards/total_composite/std": 0.16083958745002747, "reward": 0.7628706693649292, "reward_std": 0.16083957254886627, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10630126297473907, "sampling/sampling_logp_difference/max": 1.2814911603927612, "sampling/importance_sampling_ratio/min": 0.2776229977607727, "sampling/importance_sampling_ratio/mean": 0.9845349192619324, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4160885140299797, "clip_ratio/low_mean": 0.04175266623497009, "clip_ratio/low_min": 0.04175266623497009, "clip_ratio/high_mean": 0.045573493000119925, "clip_ratio/high_max": 0.045573493000119925, "clip_ratio/region_mean": 0.08732615923509002, "reward_total_mean": 0.7628706693649292, "reward_meter_mean": 0.9682329893112183, "reward_meter_std": 0.044429961591959, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.7628706693649292, "reward_total_composite_std": 0.16083958745002747} {"timestamp_utc": "2026-04-13T10:39:02Z", "mode": "train", "global_step": 1328, "epoch": 0.13340030135610245, "loss": 0.0434, "grad_norm": 11.687251091003418, "learning_rate": 5.978787878787879e-06, "num_tokens": 2346047.0, "completions/mean_length": 46.75, "completions/min_length": 43.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.75, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.6765961647033691, "rewards/meter/std": 0.35124412178993225, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.989249587059021, "rewards/repeat_soft/std": 0.014732363633811474, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.10392305999994278, "rewards/total_composite/mean": 0.5615547299385071, "rewards/total_composite/std": 0.13680243492126465, "reward": 0.5615547299385071, "reward_std": 0.13680243492126465, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1515999138355255, "sampling/sampling_logp_difference/max": 1.685828685760498, "sampling/importance_sampling_ratio/min": 0.18529081344604492, "sampling/importance_sampling_ratio/mean": 1.0011440515518188, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.849358819425106, "clip_ratio/low_mean": 0.04327042028307915, "clip_ratio/low_min": 0.04327042028307915, "clip_ratio/high_mean": 0.06864966731518507, "clip_ratio/high_max": 0.06864966731518507, "clip_ratio/region_mean": 0.11192008759826422, "reward_total_mean": 0.5615547299385071, "reward_meter_mean": 0.6765961647033691, "reward_meter_std": 0.35124412178993225, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.989249587059021, "reward_repeat_soft_std": 0.014732363633811474, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.10392305999994278, "reward_total_composite_mean": 0.5615547299385071, "reward_total_composite_std": 0.13680243492126465} {"timestamp_utc": "2026-04-13T10:39:08Z", "mode": "train", "global_step": 1329, "epoch": 0.13350075339025616, "loss": -0.0067, "grad_norm": 15.916736602783203, "learning_rate": 5.975757575757576e-06, "num_tokens": 2347335.0, "completions/mean_length": 23.0, "completions/min_length": 19.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.0, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.6458486318588257, "rewards/meter/std": 0.4809549152851105, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9220465421676636, "rewards/repeat_soft/std": 0.06529001891613007, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.5218911170959473, "rewards/total_composite/std": 0.14064864814281464, "reward": 0.5218911170959473, "reward_std": 0.14064864814281464, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1282486766576767, "sampling/sampling_logp_difference/max": 1.4058802127838135, "sampling/importance_sampling_ratio/min": 0.24515117704868317, "sampling/importance_sampling_ratio/mean": 0.9940204620361328, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8468829244375229, "clip_ratio/low_mean": 0.02794258389621973, "clip_ratio/low_min": 0.02794258389621973, "clip_ratio/high_mean": 0.1299373060464859, "clip_ratio/high_max": 0.1299373060464859, "clip_ratio/region_mean": 0.15787988994270563, "reward_total_mean": 0.5218911170959473, "reward_meter_mean": 0.6458486318588257, "reward_meter_std": 0.4809549152851105, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9220465421676636, "reward_repeat_soft_std": 0.06529001891613007, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.5218911170959473, "reward_total_composite_std": 0.14064864814281464} {"timestamp_utc": "2026-04-13T10:39:20Z", "mode": "train", "global_step": 1330, "epoch": 0.13360120542440984, "loss": -0.0218, "grad_norm": 4.435325622558594, "learning_rate": 5.972727272727274e-06, "num_tokens": 2348683.0, "completions/mean_length": 86.5, "completions/min_length": 16.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 25.71428680419922, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.47142189741134644, "rewards/meter/std": 0.3101181983947754, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.4629100561141968, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9667688608169556, "rewards/repeat_soft/std": 0.011865193955600262, "rewards/judge_quality/mean": 0.39625000953674316, "rewards/judge_quality/std": 0.14029940962791443, "rewards/total_composite/mean": 0.40942153334617615, "rewards/total_composite/std": 0.20781448483467102, "reward": 0.40942153334617615, "reward_std": 0.20781448483467102, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13891050219535828, "sampling/sampling_logp_difference/max": 0.9984486103057861, "sampling/importance_sampling_ratio/min": 0.3684506118297577, "sampling/importance_sampling_ratio/mean": 1.0111424922943115, "sampling/importance_sampling_ratio/max": 1.5134029388427734, "entropy": 0.8866551592946053, "clip_ratio/low_mean": 0.04548611119389534, "clip_ratio/low_min": 0.04548611119389534, "clip_ratio/high_mean": 0.1330641247332096, "clip_ratio/high_max": 0.1330641247332096, "clip_ratio/region_mean": 0.17855023592710495, "reward_total_mean": 0.40942153334617615, "reward_meter_mean": 0.47142189741134644, "reward_meter_std": 0.3101181983947754, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.4629100561141968, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9667688608169556, "reward_repeat_soft_std": 0.011865193955600262, "reward_judge_quality_mean": 0.39625000953674316, "reward_judge_quality_std": 0.14029940962791443, "reward_total_composite_mean": 0.40942153334617615, "reward_total_composite_std": 0.20781448483467102} {"timestamp_utc": "2026-04-13T10:39:27Z", "mode": "train", "global_step": 1331, "epoch": 0.13370165745856355, "loss": 0.0097, "grad_norm": 16.48143768310547, "learning_rate": 5.96969696969697e-06, "num_tokens": 2350438.0, "completions/mean_length": 63.375, "completions/min_length": 50.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.375, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.40556466579437256, "rewards/meter/std": 0.26941341161727905, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.1035098284482956, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9321507215499878, "rewards/repeat_soft/std": 0.04556303098797798, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4255417585372925, "rewards/total_composite/std": 0.07310432940721512, "reward": 0.4255417585372925, "reward_std": 0.07310432195663452, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.150555819272995, "sampling/sampling_logp_difference/max": 1.848578929901123, "sampling/importance_sampling_ratio/min": 0.1574607789516449, "sampling/importance_sampling_ratio/mean": 1.005539894104004, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6611043587327003, "clip_ratio/low_mean": 0.052912051789462566, "clip_ratio/low_min": 0.052912051789462566, "clip_ratio/high_mean": 0.06353189051151276, "clip_ratio/high_max": 0.06353189051151276, "clip_ratio/region_mean": 0.11644394230097532, "reward_total_mean": 0.4255417585372925, "reward_meter_mean": 0.40556466579437256, "reward_meter_std": 0.26941341161727905, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.1035098284482956, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9321507215499878, "reward_repeat_soft_std": 0.04556303098797798, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4255417585372925, "reward_total_composite_std": 0.07310432940721512} {"timestamp_utc": "2026-04-13T10:39:33Z", "mode": "train", "global_step": 1332, "epoch": 0.13380210949271723, "loss": -0.0754, "grad_norm": 17.2110652923584, "learning_rate": 5.966666666666667e-06, "num_tokens": 2352129.0, "completions/mean_length": 35.375, "completions/min_length": 31.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.375, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.314901202917099, "rewards/meter/std": 0.3396137058734894, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8801974058151245, "rewards/repeat_soft/std": 0.0764533057808876, "rewards/judge_quality/mean": 0.4937500059604645, "rewards/judge_quality/std": 0.1728696972131729, "rewards/total_composite/mean": 0.4472670555114746, "rewards/total_composite/std": 0.14360350370407104, "reward": 0.4472670555114746, "reward_std": 0.14360350370407104, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1391085535287857, "sampling/sampling_logp_difference/max": 2.2771387100219727, "sampling/importance_sampling_ratio/min": 0.10257729142904282, "sampling/importance_sampling_ratio/mean": 1.01146399974823, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7420665435492992, "clip_ratio/low_mean": 0.0835973508656025, "clip_ratio/low_min": 0.0835973508656025, "clip_ratio/high_mean": 0.0518292672932148, "clip_ratio/high_max": 0.0518292672932148, "clip_ratio/region_mean": 0.1354266181588173, "reward_total_mean": 0.4472670555114746, "reward_meter_mean": 0.314901202917099, "reward_meter_std": 0.3396137058734894, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8801974058151245, "reward_repeat_soft_std": 0.0764533057808876, "reward_judge_quality_mean": 0.4937500059604645, "reward_judge_quality_std": 0.1728696972131729, "reward_total_composite_mean": 0.4472670555114746, "reward_total_composite_std": 0.14360350370407104} {"timestamp_utc": "2026-04-13T10:39:44Z", "mode": "train", "global_step": 1333, "epoch": 0.1339025615268709, "loss": -0.1558, "grad_norm": 3.7890119552612305, "learning_rate": 5.963636363636364e-06, "num_tokens": 2354379.0, "completions/mean_length": 152.25, "completions/min_length": 66.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 100.85714721679688, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 120.0, "rewards/meter/mean": 0.2900828421115875, "rewards/meter/std": 0.2821356952190399, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.10690449178218842, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9511439800262451, "rewards/repeat_soft/std": 0.051468927413225174, "rewards/judge_quality/mean": 0.4049999713897705, "rewards/judge_quality/std": 0.19442224502563477, "rewards/total_composite/mean": 0.37064436078071594, "rewards/total_composite/std": 0.18591031432151794, "reward": 0.37064436078071594, "reward_std": 0.18591031432151794, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1486016809940338, "sampling/sampling_logp_difference/max": 1.7037146091461182, "sampling/importance_sampling_ratio/min": 0.18200618028640747, "sampling/importance_sampling_ratio/mean": 1.001570701599121, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7631268575787544, "clip_ratio/low_mean": 0.07168418262153864, "clip_ratio/low_min": 0.07168418262153864, "clip_ratio/high_mean": 0.061170581728219986, "clip_ratio/high_max": 0.061170581728219986, "clip_ratio/region_mean": 0.13285476434975863, "reward_total_mean": 0.37064436078071594, "reward_meter_mean": 0.2900828421115875, "reward_meter_std": 0.2821356952190399, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.10690449178218842, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9511439800262451, "reward_repeat_soft_std": 0.051468927413225174, "reward_judge_quality_mean": 0.4049999713897705, "reward_judge_quality_std": 0.19442224502563477, "reward_total_composite_mean": 0.37064436078071594, "reward_total_composite_std": 0.18591031432151794} {"timestamp_utc": "2026-04-13T10:39:51Z", "mode": "train", "global_step": 1334, "epoch": 0.13400301356102462, "loss": 0.0384, "grad_norm": 10.282875061035156, "learning_rate": 5.960606060606061e-06, "num_tokens": 2356568.0, "completions/mean_length": 80.625, "completions/min_length": 72.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.625, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.6107769012451172, "rewards/meter/std": 0.34543493390083313, "rewards/count_adherence/mean": 0.8958333134651184, "rewards/count_adherence/std": 0.08625820279121399, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8274047374725342, "rewards/repeat_soft/std": 0.05298303812742233, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.4572588801383972, "rewards/total_composite/std": 0.09525513648986816, "reward": 0.4572588801383972, "reward_std": 0.09525512158870697, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11038250476121902, "sampling/sampling_logp_difference/max": 1.768733024597168, "sampling/importance_sampling_ratio/min": 0.1705489307641983, "sampling/importance_sampling_ratio/mean": 1.0168070793151855, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5664704628288746, "clip_ratio/low_mean": 0.06878686603158712, "clip_ratio/low_min": 0.06878686603158712, "clip_ratio/high_mean": 0.02347811870276928, "clip_ratio/high_max": 0.02347811870276928, "clip_ratio/region_mean": 0.0922649847343564, "reward_total_mean": 0.4572588801383972, "reward_meter_mean": 0.6107769012451172, "reward_meter_std": 0.34543493390083313, "reward_count_adherence_mean": 0.8958333134651184, "reward_count_adherence_std": 0.08625820279121399, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8274047374725342, "reward_repeat_soft_std": 0.05298303812742233, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.4572588801383972, "reward_total_composite_std": 0.09525513648986816} {"timestamp_utc": "2026-04-13T10:40:03Z", "mode": "train", "global_step": 1335, "epoch": 0.1341034655951783, "loss": -0.0998, "grad_norm": 5.83372163772583, "learning_rate": 5.9575757575757575e-06, "num_tokens": 2358608.0, "completions/mean_length": 126.0, "completions/min_length": 59.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 70.85714721679688, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.3127872347831726, "rewards/meter/std": 0.3247506320476532, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.1035098284482956, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.967047393321991, "rewards/repeat_soft/std": 0.022363485768437386, "rewards/judge_quality/mean": 0.5074999928474426, "rewards/judge_quality/std": 0.2748376131057739, "rewards/total_composite/mean": 0.42580604553222656, "rewards/total_composite/std": 0.24114061892032623, "reward": 0.42580604553222656, "reward_std": 0.24114063382148743, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15268826484680176, "sampling/sampling_logp_difference/max": 2.7636923789978027, "sampling/importance_sampling_ratio/min": 0.06305850297212601, "sampling/importance_sampling_ratio/mean": 1.010711669921875, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5862419381737709, "clip_ratio/low_mean": 0.07036514859646559, "clip_ratio/low_min": 0.07036514859646559, "clip_ratio/high_mean": 0.049692231230437756, "clip_ratio/high_max": 0.049692231230437756, "clip_ratio/region_mean": 0.12005737982690334, "reward_total_mean": 0.42580604553222656, "reward_meter_mean": 0.3127872347831726, "reward_meter_std": 0.3247506320476532, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.1035098284482956, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.967047393321991, "reward_repeat_soft_std": 0.022363485768437386, "reward_judge_quality_mean": 0.5074999928474426, "reward_judge_quality_std": 0.2748376131057739, "reward_total_composite_mean": 0.42580604553222656, "reward_total_composite_std": 0.24114061892032623} {"timestamp_utc": "2026-04-13T10:40:10Z", "mode": "train", "global_step": 1336, "epoch": 0.13420391762933198, "loss": 0.0854, "grad_norm": 9.80910873413086, "learning_rate": 5.954545454545455e-06, "num_tokens": 2360746.0, "completions/mean_length": 83.25, "completions/min_length": 67.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 83.25, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.8006072640419006, "rewards/meter/std": 0.3069312870502472, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9120126962661743, "rewards/repeat_soft/std": 0.04998166114091873, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.5647725462913513, "rewards/total_composite/std": 0.11622896045446396, "reward": 0.5647725462913513, "reward_std": 0.11622896045446396, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13410262763500214, "sampling/sampling_logp_difference/max": 2.931978702545166, "sampling/importance_sampling_ratio/min": 0.05329148471355438, "sampling/importance_sampling_ratio/mean": 1.0088557004928589, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7084437683224678, "clip_ratio/low_mean": 0.046615186147391796, "clip_ratio/low_min": 0.046615186147391796, "clip_ratio/high_mean": 0.08461390342563391, "clip_ratio/high_max": 0.08461390342563391, "clip_ratio/region_mean": 0.1312290895730257, "reward_total_mean": 0.5647725462913513, "reward_meter_mean": 0.8006072640419006, "reward_meter_std": 0.3069312870502472, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9120126962661743, "reward_repeat_soft_std": 0.04998166114091873, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.5647725462913513, "reward_total_composite_std": 0.11622896045446396} {"timestamp_utc": "2026-04-13T10:40:16Z", "mode": "train", "global_step": 1337, "epoch": 0.1343043696634857, "loss": 0.0476, "grad_norm": 15.51915454864502, "learning_rate": 5.951515151515151e-06, "num_tokens": 2362513.0, "completions/mean_length": 44.875, "completions/min_length": 33.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.875, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.710849404335022, "rewards/meter/std": 0.33453720808029175, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9896112680435181, "rewards/repeat_soft/std": 0.012233623303472996, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.17136012017726898, "rewards/total_composite/mean": 0.5692287683486938, "rewards/total_composite/std": 0.0953088030219078, "reward": 0.5692287683486938, "reward_std": 0.09530878812074661, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1775110512971878, "sampling/sampling_logp_difference/max": 1.4458134174346924, "sampling/importance_sampling_ratio/min": 0.23555439710617065, "sampling/importance_sampling_ratio/mean": 0.9923685193061829, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1398600190877914, "clip_ratio/low_mean": 0.025348837487399578, "clip_ratio/low_min": 0.025348837487399578, "clip_ratio/high_mean": 0.1504751518368721, "clip_ratio/high_max": 0.1504751518368721, "clip_ratio/region_mean": 0.17582398932427168, "reward_total_mean": 0.5692287683486938, "reward_meter_mean": 0.710849404335022, "reward_meter_std": 0.33453720808029175, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9896112680435181, "reward_repeat_soft_std": 0.012233623303472996, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.17136012017726898, "reward_total_composite_mean": 0.5692287683486938, "reward_total_composite_std": 0.0953088030219078} {"timestamp_utc": "2026-04-13T10:40:22Z", "mode": "train", "global_step": 1338, "epoch": 0.13440482169763937, "loss": 0.0401, "grad_norm": 15.774927139282227, "learning_rate": 5.948484848484849e-06, "num_tokens": 2364045.0, "completions/mean_length": 37.5, "completions/min_length": 33.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.5, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.8714721202850342, "rewards/meter/std": 0.30579236149787903, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.878113865852356, "rewards/repeat_soft/std": 0.045858755707740784, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.6930241584777832, "rewards/total_composite/std": 0.19361813366413116, "reward": 0.6930241584777832, "reward_std": 0.19361810386180878, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12548218667507172, "sampling/sampling_logp_difference/max": 1.1062960624694824, "sampling/importance_sampling_ratio/min": 0.40296289324760437, "sampling/importance_sampling_ratio/mean": 1.041352391242981, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8229580894112587, "clip_ratio/low_mean": 0.0958860320970416, "clip_ratio/low_min": 0.0958860320970416, "clip_ratio/high_mean": 0.054523564875125885, "clip_ratio/high_max": 0.054523564875125885, "clip_ratio/region_mean": 0.1504095969721675, "reward_total_mean": 0.6930241584777832, "reward_meter_mean": 0.8714721202850342, "reward_meter_std": 0.30579236149787903, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.878113865852356, "reward_repeat_soft_std": 0.045858755707740784, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.6930241584777832, "reward_total_composite_std": 0.19361813366413116} {"timestamp_utc": "2026-04-13T10:40:28Z", "mode": "train", "global_step": 1339, "epoch": 0.13450527373179308, "loss": 0.0041, "grad_norm": 14.936269760131836, "learning_rate": 5.9454545454545465e-06, "num_tokens": 2365582.0, "completions/mean_length": 39.125, "completions/min_length": 35.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.125, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.7665152549743652, "rewards/meter/std": 0.3787933588027954, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9921016097068787, "rewards/repeat_soft/std": 0.007253531366586685, "rewards/judge_quality/mean": 0.5575000047683716, "rewards/judge_quality/std": 0.19955310225486755, "rewards/total_composite/mean": 0.6462671756744385, "rewards/total_composite/std": 0.19861456751823425, "reward": 0.6462671756744385, "reward_std": 0.19861455261707306, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14262601733207703, "sampling/sampling_logp_difference/max": 1.219222068786621, "sampling/importance_sampling_ratio/min": 0.31959930062294006, "sampling/importance_sampling_ratio/mean": 1.0433692932128906, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9027925133705139, "clip_ratio/low_mean": 0.08850973658263683, "clip_ratio/low_min": 0.08850973658263683, "clip_ratio/high_mean": 0.059057971462607384, "clip_ratio/high_max": 0.059057971462607384, "clip_ratio/region_mean": 0.14756770804524422, "reward_total_mean": 0.6462671756744385, "reward_meter_mean": 0.7665152549743652, "reward_meter_std": 0.3787933588027954, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9921016097068787, "reward_repeat_soft_std": 0.007253531366586685, "reward_judge_quality_mean": 0.5575000047683716, "reward_judge_quality_std": 0.19955310225486755, "reward_total_composite_mean": 0.6462671756744385, "reward_total_composite_std": 0.19861456751823425} {"timestamp_utc": "2026-04-13T10:40:37Z", "mode": "train", "global_step": 1340, "epoch": 0.13460572576594676, "loss": 0.0193, "grad_norm": 7.9399895668029785, "learning_rate": 5.942424242424243e-06, "num_tokens": 2368188.0, "completions/mean_length": 117.75, "completions/min_length": 104.0, "completions/max_length": 140.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 117.75, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 140.0, "rewards/meter/mean": 0.5848919153213501, "rewards/meter/std": 0.31664523482322693, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8285818099975586, "rewards/repeat_soft/std": 0.058063607662916183, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.45062944293022156, "rewards/total_composite/std": 0.08520717173814774, "reward": 0.45062944293022156, "reward_std": 0.08520715683698654, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13698212802410126, "sampling/sampling_logp_difference/max": 2.5663671493530273, "sampling/importance_sampling_ratio/min": 0.07681409269571304, "sampling/importance_sampling_ratio/mean": 0.9988870620727539, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7696087509393692, "clip_ratio/low_mean": 0.05009881127625704, "clip_ratio/low_min": 0.05009881127625704, "clip_ratio/high_mean": 0.0640478739514947, "clip_ratio/high_max": 0.0640478739514947, "clip_ratio/region_mean": 0.11414668522775173, "reward_total_mean": 0.45062944293022156, "reward_meter_mean": 0.5848919153213501, "reward_meter_std": 0.31664523482322693, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8285818099975586, "reward_repeat_soft_std": 0.058063607662916183, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.45062944293022156, "reward_total_composite_std": 0.08520717173814774} {"timestamp_utc": "2026-04-13T10:40:43Z", "mode": "train", "global_step": 1341, "epoch": 0.13470617780010044, "loss": 0.0377, "grad_norm": 18.52684211730957, "learning_rate": 5.93939393939394e-06, "num_tokens": 2369644.0, "completions/mean_length": 25.0, "completions/min_length": 22.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.0, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.6021800637245178, "rewards/meter/std": 0.4963541626930237, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9615821838378906, "rewards/repeat_soft/std": 0.002596014179289341, "rewards/judge_quality/mean": 0.4937500059604645, "rewards/judge_quality/std": 0.1728696972131729, "rewards/total_composite/mean": 0.5509586930274963, "rewards/total_composite/std": 0.20226746797561646, "reward": 0.5509586930274963, "reward_std": 0.20226743817329407, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1375489979982376, "sampling/sampling_logp_difference/max": 1.6032088994979858, "sampling/importance_sampling_ratio/min": 0.20124968886375427, "sampling/importance_sampling_ratio/mean": 1.0113073587417603, "sampling/importance_sampling_ratio/max": 1.8520575761795044, "entropy": 0.9718766510486603, "clip_ratio/low_mean": 0.04949534125626087, "clip_ratio/low_min": 0.04949534125626087, "clip_ratio/high_mean": 0.06608553603291512, "clip_ratio/high_max": 0.06608553603291512, "clip_ratio/region_mean": 0.11558087728917599, "reward_total_mean": 0.5509586930274963, "reward_meter_mean": 0.6021800637245178, "reward_meter_std": 0.4963541626930237, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9615821838378906, "reward_repeat_soft_std": 0.002596014179289341, "reward_judge_quality_mean": 0.4937500059604645, "reward_judge_quality_std": 0.1728696972131729, "reward_total_composite_mean": 0.5509586930274963, "reward_total_composite_std": 0.20226746797561646} {"timestamp_utc": "2026-04-13T10:40:49Z", "mode": "train", "global_step": 1342, "epoch": 0.13480662983425415, "loss": -0.0059, "grad_norm": 15.406861305236816, "learning_rate": 5.936363636363637e-06, "num_tokens": 2371339.0, "completions/mean_length": 44.875, "completions/min_length": 39.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.875, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.8774193525314331, "rewards/meter/std": 0.27800363302230835, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8659125566482544, "rewards/repeat_soft/std": 0.05217598378658295, "rewards/judge_quality/mean": 0.4724999964237213, "rewards/judge_quality/std": 0.11310552060604095, "rewards/total_composite/mean": 0.5258831977844238, "rewards/total_composite/std": 0.24142670631408691, "reward": 0.5258831977844238, "reward_std": 0.24142670631408691, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12861864268779755, "sampling/sampling_logp_difference/max": 1.7518625259399414, "sampling/importance_sampling_ratio/min": 0.17345057427883148, "sampling/importance_sampling_ratio/mean": 1.0132513046264648, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8706328645348549, "clip_ratio/low_mean": 0.019702842459082603, "clip_ratio/low_min": 0.019702842459082603, "clip_ratio/high_mean": 0.10141318663954735, "clip_ratio/high_max": 0.10141318663954735, "clip_ratio/region_mean": 0.12111602909862995, "reward_total_mean": 0.5258831977844238, "reward_meter_mean": 0.8774193525314331, "reward_meter_std": 0.27800363302230835, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8659125566482544, "reward_repeat_soft_std": 0.05217598378658295, "reward_judge_quality_mean": 0.4724999964237213, "reward_judge_quality_std": 0.11310552060604095, "reward_total_composite_mean": 0.5258831977844238, "reward_total_composite_std": 0.24142670631408691} {"timestamp_utc": "2026-04-13T10:40:55Z", "mode": "train", "global_step": 1343, "epoch": 0.13490708186840783, "loss": 0.0864, "grad_norm": 14.79797649383545, "learning_rate": 5.933333333333335e-06, "num_tokens": 2373021.0, "completions/mean_length": 40.25, "completions/min_length": 34.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.25, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.8833025693893433, "rewards/meter/std": 0.2559974193572998, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8911500573158264, "rewards/repeat_soft/std": 0.05381282418966293, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.5612891912460327, "rewards/total_composite/std": 0.06968934088945389, "reward": 0.5612891912460327, "reward_std": 0.06968934834003448, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14629581570625305, "sampling/sampling_logp_difference/max": 1.823883056640625, "sampling/importance_sampling_ratio/min": 0.1613978147506714, "sampling/importance_sampling_ratio/mean": 1.02433443069458, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7586573287844658, "clip_ratio/low_mean": 0.038345410488545895, "clip_ratio/low_min": 0.038345410488545895, "clip_ratio/high_mean": 0.09234451642259955, "clip_ratio/high_max": 0.09234451642259955, "clip_ratio/region_mean": 0.13068992691114545, "reward_total_mean": 0.5612891912460327, "reward_meter_mean": 0.8833025693893433, "reward_meter_std": 0.2559974193572998, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8911500573158264, "reward_repeat_soft_std": 0.05381282418966293, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.5612891912460327, "reward_total_composite_std": 0.06968934088945389} {"timestamp_utc": "2026-04-13T10:41:01Z", "mode": "train", "global_step": 1344, "epoch": 0.13500753390256154, "loss": 0.0742, "grad_norm": 11.261421203613281, "learning_rate": 5.93030303030303e-06, "num_tokens": 2374570.0, "completions/mean_length": 41.625, "completions/min_length": 38.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.625, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.8662797808647156, "rewards/meter/std": 0.17917276918888092, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9751627445220947, "rewards/repeat_soft/std": 0.04239398241043091, "rewards/judge_quality/mean": 0.5849999785423279, "rewards/judge_quality/std": 0.21967509388923645, "rewards/total_composite/mean": 0.6761919260025024, "rewards/total_composite/std": 0.1401374489068985, "reward": 0.6761919260025024, "reward_std": 0.1401374489068985, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13689248263835907, "sampling/sampling_logp_difference/max": 2.1067726612091064, "sampling/importance_sampling_ratio/min": 0.12162987142801285, "sampling/importance_sampling_ratio/mean": 1.0012785196304321, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7067522257566452, "clip_ratio/low_mean": 0.07602102914825082, "clip_ratio/low_min": 0.07602102914825082, "clip_ratio/high_mean": 0.05430820258334279, "clip_ratio/high_max": 0.05430820258334279, "clip_ratio/region_mean": 0.1303292317315936, "reward_total_mean": 0.6761919260025024, "reward_meter_mean": 0.8662797808647156, "reward_meter_std": 0.17917276918888092, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9751627445220947, "reward_repeat_soft_std": 0.04239398241043091, "reward_judge_quality_mean": 0.5849999785423279, "reward_judge_quality_std": 0.21967509388923645, "reward_total_composite_mean": 0.6761919260025024, "reward_total_composite_std": 0.1401374489068985} {"timestamp_utc": "2026-04-13T10:41:08Z", "mode": "train", "global_step": 1345, "epoch": 0.13510798593671522, "loss": 0.0363, "grad_norm": 9.135940551757812, "learning_rate": 5.927272727272728e-06, "num_tokens": 2376381.0, "completions/mean_length": 55.375, "completions/min_length": 49.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.375, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9781638383865356, "rewards/meter/std": 0.006795698311179876, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7677721977233887, "rewards/repeat_soft/std": 0.07447344064712524, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5822045803070068, "rewards/total_composite/std": 0.010020908899605274, "reward": 0.5822045803070068, "reward_std": 0.010020908899605274, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08400870859622955, "sampling/sampling_logp_difference/max": 1.305046558380127, "sampling/importance_sampling_ratio/min": 0.27115991711616516, "sampling/importance_sampling_ratio/mean": 1.0251963138580322, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.49729280546307564, "clip_ratio/low_mean": 0.04706370458006859, "clip_ratio/low_min": 0.04706370458006859, "clip_ratio/high_mean": 0.035152471624314785, "clip_ratio/high_max": 0.035152471624314785, "clip_ratio/region_mean": 0.08221617620438337, "reward_total_mean": 0.5822045803070068, "reward_meter_mean": 0.9781638383865356, "reward_meter_std": 0.006795698311179876, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7677721977233887, "reward_repeat_soft_std": 0.07447344064712524, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5822045803070068, "reward_total_composite_std": 0.010020908899605274} {"timestamp_utc": "2026-04-13T10:41:14Z", "mode": "train", "global_step": 1346, "epoch": 0.1352084379708689, "loss": 0.0451, "grad_norm": 15.326791763305664, "learning_rate": 5.924242424242425e-06, "num_tokens": 2378465.0, "completions/mean_length": 80.5, "completions/min_length": 75.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.5, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.8029333353042603, "rewards/meter/std": 0.30681586265563965, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8098430633544922, "rewards/repeat_soft/std": 0.06636009365320206, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.5424083471298218, "rewards/total_composite/std": 0.07780204713344574, "reward": 0.5424083471298218, "reward_std": 0.07780203223228455, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10905911773443222, "sampling/sampling_logp_difference/max": 1.425297737121582, "sampling/importance_sampling_ratio/min": 0.2404368668794632, "sampling/importance_sampling_ratio/mean": 1.0072455406188965, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6222714483737946, "clip_ratio/low_mean": 0.03178161010146141, "clip_ratio/low_min": 0.03178161010146141, "clip_ratio/high_mean": 0.06335204932838678, "clip_ratio/high_max": 0.06335204932838678, "clip_ratio/region_mean": 0.0951336594298482, "reward_total_mean": 0.5424083471298218, "reward_meter_mean": 0.8029333353042603, "reward_meter_std": 0.30681586265563965, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8098430633544922, "reward_repeat_soft_std": 0.06636009365320206, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.5424083471298218, "reward_total_composite_std": 0.07780204713344574} {"timestamp_utc": "2026-04-13T10:41:20Z", "mode": "train", "global_step": 1347, "epoch": 0.1353088900050226, "loss": 0.0868, "grad_norm": 13.039422035217285, "learning_rate": 5.921212121212122e-06, "num_tokens": 2380197.0, "completions/mean_length": 44.5, "completions/min_length": 38.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.5, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.8421345353126526, "rewards/meter/std": 0.22694624960422516, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8642581701278687, "rewards/repeat_soft/std": 0.07040487229824066, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.17136012017726898, "rewards/total_composite/mean": 0.6084120869636536, "rewards/total_composite/std": 0.14150801301002502, "reward": 0.6084120869636536, "reward_std": 0.14150801301002502, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13987328112125397, "sampling/sampling_logp_difference/max": 1.5131187438964844, "sampling/importance_sampling_ratio/min": 0.2202221006155014, "sampling/importance_sampling_ratio/mean": 0.9914326071739197, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7605642080307007, "clip_ratio/low_mean": 0.05355572560802102, "clip_ratio/low_min": 0.05355572560802102, "clip_ratio/high_mean": 0.06820268929004669, "clip_ratio/high_max": 0.06820268929004669, "clip_ratio/region_mean": 0.12175841489806771, "reward_total_mean": 0.6084120869636536, "reward_meter_mean": 0.8421345353126526, "reward_meter_std": 0.22694624960422516, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8642581701278687, "reward_repeat_soft_std": 0.07040487229824066, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.17136012017726898, "reward_total_composite_mean": 0.6084120869636536, "reward_total_composite_std": 0.14150801301002502} {"timestamp_utc": "2026-04-13T10:41:26Z", "mode": "train", "global_step": 1348, "epoch": 0.1354093420391763, "loss": -0.001, "grad_norm": 20.71615219116211, "learning_rate": 5.9181818181818184e-06, "num_tokens": 2381548.0, "completions/mean_length": 18.875, "completions/min_length": 14.0, "completions/max_length": 23.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 18.875, "completions/min_terminated_length": 14.0, "completions/max_terminated_length": 23.0, "rewards/meter/mean": 0.9922042489051819, "rewards/meter/std": 0.0061484333127737045, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.915888786315918, "rewards/repeat_soft/std": 0.06378740072250366, "rewards/judge_quality/mean": 0.4137499928474426, "rewards/judge_quality/std": 0.06781013309955597, "rewards/total_composite/mean": 0.6042660474777222, "rewards/total_composite/std": 0.04260183125734329, "reward": 0.6042660474777222, "reward_std": 0.04260184243321419, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13716576993465424, "sampling/sampling_logp_difference/max": 0.8954238891601562, "sampling/importance_sampling_ratio/min": 0.4084344208240509, "sampling/importance_sampling_ratio/mean": 1.0150609016418457, "sampling/importance_sampling_ratio/max": 1.8645116090774536, "entropy": 0.9576961472630501, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0702574853785336, "clip_ratio/high_max": 0.0702574853785336, "clip_ratio/region_mean": 0.0702574853785336, "reward_total_mean": 0.6042660474777222, "reward_meter_mean": 0.9922042489051819, "reward_meter_std": 0.0061484333127737045, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.915888786315918, "reward_repeat_soft_std": 0.06378740072250366, "reward_judge_quality_mean": 0.4137499928474426, "reward_judge_quality_std": 0.06781013309955597, "reward_total_composite_mean": 0.6042660474777222, "reward_total_composite_std": 0.04260183125734329} {"timestamp_utc": "2026-04-13T10:41:32Z", "mode": "train", "global_step": 1349, "epoch": 0.13550979407333, "loss": 0.0282, "grad_norm": 11.147529602050781, "learning_rate": 5.915151515151516e-06, "num_tokens": 2383138.0, "completions/mean_length": 47.75, "completions/min_length": 43.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.75, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.8343709707260132, "rewards/meter/std": 0.3385268747806549, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9651442766189575, "rewards/repeat_soft/std": 0.024144303053617477, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.596796989440918, "rewards/total_composite/std": 0.15304796397686005, "reward": 0.596796989440918, "reward_std": 0.15304794907569885, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1502106934785843, "sampling/sampling_logp_difference/max": 1.7638282775878906, "sampling/importance_sampling_ratio/min": 0.17138749361038208, "sampling/importance_sampling_ratio/mean": 1.0072975158691406, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9210467338562012, "clip_ratio/low_mean": 0.05865122191607952, "clip_ratio/low_min": 0.05865122191607952, "clip_ratio/high_mean": 0.0917295441031456, "clip_ratio/high_max": 0.0917295441031456, "clip_ratio/region_mean": 0.15038076601922512, "reward_total_mean": 0.596796989440918, "reward_meter_mean": 0.8343709707260132, "reward_meter_std": 0.3385268747806549, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9651442766189575, "reward_repeat_soft_std": 0.024144303053617477, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.596796989440918, "reward_total_composite_std": 0.15304796397686005} {"timestamp_utc": "2026-04-13T10:41:39Z", "mode": "train", "global_step": 1350, "epoch": 0.13561024610748368, "loss": 0.0367, "grad_norm": 11.28422737121582, "learning_rate": 5.912121212121212e-06, "num_tokens": 2384844.0, "completions/mean_length": 46.25, "completions/min_length": 40.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.25, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.4069119691848755, "rewards/meter/std": 0.37338826060295105, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.986085832118988, "rewards/repeat_soft/std": 0.01686898060142994, "rewards/judge_quality/mean": 0.7450000047683716, "rewards/judge_quality/std": 0.16690459847450256, "rewards/total_composite/mean": 0.5481597185134888, "rewards/total_composite/std": 0.2011103332042694, "reward": 0.5481597185134888, "reward_std": 0.20111030340194702, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13221058249473572, "sampling/sampling_logp_difference/max": 1.7689342498779297, "sampling/importance_sampling_ratio/min": 0.17051461338996887, "sampling/importance_sampling_ratio/mean": 0.9992238879203796, "sampling/importance_sampling_ratio/max": 1.8740284442901611, "entropy": 0.9551234468817711, "clip_ratio/low_mean": 0.08280645590275526, "clip_ratio/low_min": 0.08280645590275526, "clip_ratio/high_mean": 0.06803585588932037, "clip_ratio/high_max": 0.06803585588932037, "clip_ratio/region_mean": 0.15084231179207563, "reward_total_mean": 0.5481597185134888, "reward_meter_mean": 0.4069119691848755, "reward_meter_std": 0.37338826060295105, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.986085832118988, "reward_repeat_soft_std": 0.01686898060142994, "reward_judge_quality_mean": 0.7450000047683716, "reward_judge_quality_std": 0.16690459847450256, "reward_total_composite_mean": 0.5481597185134888, "reward_total_composite_std": 0.2011103332042694} {"timestamp_utc": "2026-04-13T10:42:16Z", "mode": "eval", "global_step": 1350, "epoch": 0.13561024610748368, "eval_loss": NaN, "eval_runtime": 36.8843, "eval_samples_per_second": 2.169, "eval_steps_per_second": 0.271, "eval_num_tokens": 2384844.0, "eval_completions/mean_length": 64.8625, "eval_completions/min_length": 29.9, "eval_completions/max_length": 107.8, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 64.8625, "eval_completions/min_terminated_length": 29.9, "eval_completions/max_terminated_length": 107.8, "eval_rewards/meter/mean": 0.6510038495063781, "eval_rewards/meter/std": 0.369542495906353, "eval_rewards/count_adherence/mean": 0.9585416555404663, "eval_rewards/count_adherence/std": 0.07452610954642296, "eval_rewards/hard_gate/mean": 1.0, "eval_rewards/hard_gate/std": 0.0, "eval_rewards/repeat_soft/mean": 0.881782990694046, "eval_rewards/repeat_soft/std": 0.09381132461130619, "eval_rewards/judge_quality/mean": 0.467249995470047, "eval_rewards/judge_quality/std": 0.16026114895939828, "eval_rewards/total_composite/mean": 0.5154408305883408, "eval_rewards/total_composite/std": 0.1306465409696102, "eval_reward": 0.5154408305883408, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.06699008122086525, "eval_sampling/sampling_logp_difference/max": 1.0151650428771972, "eval_sampling/importance_sampling_ratio/min": 0.3660672038793564, "eval_sampling/importance_sampling_ratio/mean": 1.0170485734939576, "eval_sampling/importance_sampling_ratio/max": 1.408859872817993, "eval_entropy": 0.7701298058032989, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5154408305883408, "eval_reward_meter_mean": 0.6510038495063781, "eval_reward_meter_std": 0.369542495906353, "eval_reward_count_adherence_mean": 0.9585416555404663, "eval_reward_count_adherence_std": 0.07452610954642296, "eval_reward_hard_gate_mean": 1.0, "eval_reward_hard_gate_std": 0.0, "eval_reward_repeat_soft_mean": 0.881782990694046, "eval_reward_repeat_soft_std": 0.09381132461130619, "eval_reward_judge_quality_mean": 0.467249995470047, "eval_reward_judge_quality_std": 0.16026114895939828, "eval_reward_total_composite_mean": 0.5154408305883408, "eval_reward_total_composite_std": 0.1306465409696102} {"timestamp_utc": "2026-04-13T10:42:25Z", "mode": "train", "global_step": 1351, "epoch": 0.13571069814163736, "loss": 0.1548, "grad_norm": 17.512495040893555, "learning_rate": 5.90909090909091e-06, "num_tokens": 2386208.0, "completions/mean_length": 21.5, "completions/min_length": 18.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.5, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.5794750452041626, "rewards/meter/std": 0.4747667908668518, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9476755857467651, "rewards/repeat_soft/std": 0.041929665952920914, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.14320313930511475, "rewards/total_composite/mean": 0.4657644033432007, "rewards/total_composite/std": 0.12241081148386002, "reward": 0.4657644033432007, "reward_std": 0.12241081148386002, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12897294759750366, "sampling/sampling_logp_difference/max": 0.9636764526367188, "sampling/importance_sampling_ratio/min": 0.38148778676986694, "sampling/importance_sampling_ratio/mean": 0.9785038828849792, "sampling/importance_sampling_ratio/max": 1.5874792337417603, "entropy": 0.8263326734304428, "clip_ratio/low_mean": 0.06844179518520832, "clip_ratio/low_min": 0.06844179518520832, "clip_ratio/high_mean": 0.048245614394545555, "clip_ratio/high_max": 0.048245614394545555, "clip_ratio/region_mean": 0.11668740957975388, "reward_total_mean": 0.4657644033432007, "reward_meter_mean": 0.5794750452041626, "reward_meter_std": 0.4747667908668518, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9476755857467651, "reward_repeat_soft_std": 0.041929665952920914, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.14320313930511475, "reward_total_composite_mean": 0.4657644033432007, "reward_total_composite_std": 0.12241081148386002} {"timestamp_utc": "2026-04-13T10:42:31Z", "mode": "train", "global_step": 1352, "epoch": 0.13581115017579107, "loss": -0.0074, "grad_norm": 11.55992603302002, "learning_rate": 5.906060606060607e-06, "num_tokens": 2387877.0, "completions/mean_length": 44.625, "completions/min_length": 36.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.625, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.32143399119377136, "rewards/meter/std": 0.4152473509311676, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9601118564605713, "rewards/repeat_soft/std": 0.023664511740207672, "rewards/judge_quality/mean": 0.5237500071525574, "rewards/judge_quality/std": 0.19078317284584045, "rewards/total_composite/mean": 0.4557591676712036, "rewards/total_composite/std": 0.14588448405265808, "reward": 0.4557591676712036, "reward_std": 0.14588449895381927, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14616696536540985, "sampling/sampling_logp_difference/max": 1.1126775741577148, "sampling/importance_sampling_ratio/min": 0.3286777138710022, "sampling/importance_sampling_ratio/mean": 1.0224692821502686, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0526707097887993, "clip_ratio/low_mean": 0.06728621479123831, "clip_ratio/low_min": 0.06728621479123831, "clip_ratio/high_mean": 0.05823863670229912, "clip_ratio/high_max": 0.05823863670229912, "clip_ratio/region_mean": 0.12552485149353743, "reward_total_mean": 0.4557591676712036, "reward_meter_mean": 0.32143399119377136, "reward_meter_std": 0.4152473509311676, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9601118564605713, "reward_repeat_soft_std": 0.023664511740207672, "reward_judge_quality_mean": 0.5237500071525574, "reward_judge_quality_std": 0.19078317284584045, "reward_total_composite_mean": 0.4557591676712036, "reward_total_composite_std": 0.14588448405265808} {"timestamp_utc": "2026-04-13T10:42:43Z", "mode": "train", "global_step": 1353, "epoch": 0.13591160220994475, "loss": -0.114, "grad_norm": 2.1518027782440186, "learning_rate": 5.903030303030304e-06, "num_tokens": 2389547.0, "completions/mean_length": 104.75, "completions/min_length": 43.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 46.57143020629883, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.595986008644104, "rewards/meter/std": 0.32510465383529663, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8903664946556091, "rewards/repeat_soft/std": 0.10740083456039429, "rewards/judge_quality/mean": 0.3387500047683716, "rewards/judge_quality/std": 0.14327171444892883, "rewards/total_composite/mean": 0.43036508560180664, "rewards/total_composite/std": 0.19207099080085754, "reward": 0.43036508560180664, "reward_std": 0.19207097589969635, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11597153544425964, "sampling/sampling_logp_difference/max": 1.6034255027770996, "sampling/importance_sampling_ratio/min": 0.20120610296726227, "sampling/importance_sampling_ratio/mean": 1.011412262916565, "sampling/importance_sampling_ratio/max": 1.964923620223999, "entropy": 0.8036029450595379, "clip_ratio/low_mean": 0.026701120659708977, "clip_ratio/low_min": 0.026701120659708977, "clip_ratio/high_mean": 0.07077431888319552, "clip_ratio/high_max": 0.07077431888319552, "clip_ratio/region_mean": 0.0974754395429045, "reward_total_mean": 0.43036508560180664, "reward_meter_mean": 0.595986008644104, "reward_meter_std": 0.32510465383529663, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8903664946556091, "reward_repeat_soft_std": 0.10740083456039429, "reward_judge_quality_mean": 0.3387500047683716, "reward_judge_quality_std": 0.14327171444892883, "reward_total_composite_mean": 0.43036508560180664, "reward_total_composite_std": 0.19207099080085754} {"timestamp_utc": "2026-04-13T10:42:52Z", "mode": "train", "global_step": 1354, "epoch": 0.13601205424409843, "loss": 0.01, "grad_norm": 12.866393089294434, "learning_rate": 5.9e-06, "num_tokens": 2391142.0, "completions/mean_length": 35.375, "completions/min_length": 33.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.375, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.8208308219909668, "rewards/meter/std": 0.25697070360183716, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9737215638160706, "rewards/repeat_soft/std": 0.025722060352563858, "rewards/judge_quality/mean": 0.47749996185302734, "rewards/judge_quality/std": 0.16263456642627716, "rewards/total_composite/mean": 0.605654776096344, "rewards/total_composite/std": 0.1339626908302307, "reward": 0.605654776096344, "reward_std": 0.13396267592906952, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11455649137496948, "sampling/sampling_logp_difference/max": 1.409034252166748, "sampling/importance_sampling_ratio/min": 0.2443791776895523, "sampling/importance_sampling_ratio/mean": 1.0094857215881348, "sampling/importance_sampling_ratio/max": 1.966202974319458, "entropy": 0.630085363984108, "clip_ratio/low_mean": 0.04261363763362169, "clip_ratio/low_min": 0.04261363763362169, "clip_ratio/high_mean": 0.07221631123684347, "clip_ratio/high_max": 0.07221631123684347, "clip_ratio/region_mean": 0.11482994887046516, "reward_total_mean": 0.605654776096344, "reward_meter_mean": 0.8208308219909668, "reward_meter_std": 0.25697070360183716, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9737215638160706, "reward_repeat_soft_std": 0.025722060352563858, "reward_judge_quality_mean": 0.47749996185302734, "reward_judge_quality_std": 0.16263456642627716, "reward_total_composite_mean": 0.605654776096344, "reward_total_composite_std": 0.1339626908302307} {"timestamp_utc": "2026-04-13T10:42:58Z", "mode": "train", "global_step": 1355, "epoch": 0.13611250627825214, "loss": 0.0844, "grad_norm": 16.96038055419922, "learning_rate": 5.8969696969696975e-06, "num_tokens": 2392560.0, "completions/mean_length": 21.25, "completions/min_length": 20.0, "completions/max_length": 23.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.25, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 23.0, "rewards/meter/mean": 0.7864716649055481, "rewards/meter/std": 0.3567924499511719, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9309078454971313, "rewards/repeat_soft/std": 0.04106544330716133, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.5947766900062561, "rewards/total_composite/std": 0.1688213050365448, "reward": 0.5947766900062561, "reward_std": 0.1688213050365448, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13679121434688568, "sampling/sampling_logp_difference/max": 1.2551536560058594, "sampling/importance_sampling_ratio/min": 0.2850320339202881, "sampling/importance_sampling_ratio/mean": 1.0153204202651978, "sampling/importance_sampling_ratio/max": 1.9045906066894531, "entropy": 0.9432927444577217, "clip_ratio/low_mean": 0.03880811156705022, "clip_ratio/low_min": 0.03880811156705022, "clip_ratio/high_mean": 0.10338203702121973, "clip_ratio/high_max": 0.10338203702121973, "clip_ratio/region_mean": 0.14219014858826995, "reward_total_mean": 0.5947766900062561, "reward_meter_mean": 0.7864716649055481, "reward_meter_std": 0.3567924499511719, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9309078454971313, "reward_repeat_soft_std": 0.04106544330716133, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.5947766900062561, "reward_total_composite_std": 0.1688213050365448} {"timestamp_utc": "2026-04-13T10:43:04Z", "mode": "train", "global_step": 1356, "epoch": 0.13621295831240582, "loss": 0.0937, "grad_norm": 17.286178588867188, "learning_rate": 5.893939393939394e-06, "num_tokens": 2394154.0, "completions/mean_length": 39.25, "completions/min_length": 33.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.25, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.33529916405677795, "rewards/meter/std": 0.3962272107601166, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9275178909301758, "rewards/repeat_soft/std": 0.05657338723540306, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.4632614552974701, "rewards/total_composite/std": 0.20912042260169983, "reward": 0.4632614552974701, "reward_std": 0.20912039279937744, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1105823665857315, "sampling/sampling_logp_difference/max": 1.0438995361328125, "sampling/importance_sampling_ratio/min": 0.38390469551086426, "sampling/importance_sampling_ratio/mean": 1.0394328832626343, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7477941028773785, "clip_ratio/low_mean": 0.0745665249414742, "clip_ratio/low_min": 0.0745665249414742, "clip_ratio/high_mean": 0.022727273404598236, "clip_ratio/high_max": 0.022727273404598236, "clip_ratio/region_mean": 0.09729379834607244, "reward_total_mean": 0.4632614552974701, "reward_meter_mean": 0.33529916405677795, "reward_meter_std": 0.3962272107601166, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9275178909301758, "reward_repeat_soft_std": 0.05657338723540306, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.4632614552974701, "reward_total_composite_std": 0.20912042260169983} {"timestamp_utc": "2026-04-13T10:43:16Z", "mode": "train", "global_step": 1357, "epoch": 0.13631341034655953, "loss": -0.1563, "grad_norm": 2.4548747539520264, "learning_rate": 5.890909090909091e-06, "num_tokens": 2396176.0, "completions/mean_length": 248.75, "completions/min_length": 81.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 90.80000305175781, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.411411315202713, "rewards/meter/std": 0.2769858241081238, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.9301238059997559, "rewards/repeat_soft/std": 0.07403453439474106, "rewards/judge_quality/mean": 0.3374999761581421, "rewards/judge_quality/std": 0.30372685194015503, "rewards/total_composite/mean": 0.29426518082618713, "rewards/total_composite/std": 0.2515621781349182, "reward": 0.29426518082618713, "reward_std": 0.2515621483325958, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12146399170160294, "sampling/sampling_logp_difference/max": 1.476247787475586, "sampling/importance_sampling_ratio/min": 0.22849343717098236, "sampling/importance_sampling_ratio/mean": 1.0232826471328735, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5180720612406731, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0682307556271553, "clip_ratio/high_max": 0.0682307556271553, "clip_ratio/region_mean": 0.0682307556271553, "reward_total_mean": 0.29426518082618713, "reward_meter_mean": 0.411411315202713, "reward_meter_std": 0.2769858241081238, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.9301238059997559, "reward_repeat_soft_std": 0.07403453439474106, "reward_judge_quality_mean": 0.3374999761581421, "reward_judge_quality_std": 0.30372685194015503, "reward_total_composite_mean": 0.29426518082618713, "reward_total_composite_std": 0.2515621781349182} {"timestamp_utc": "2026-04-13T10:43:28Z", "mode": "train", "global_step": 1358, "epoch": 0.1364138623807132, "loss": -0.1726, "grad_norm": 2.606265068054199, "learning_rate": 5.887878787878788e-06, "num_tokens": 2398224.0, "completions/mean_length": 142.0, "completions/min_length": 82.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 89.14286041259766, "completions/min_terminated_length": 82.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.5142291784286499, "rewards/meter/std": 0.27461180090904236, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8774094581604004, "rewards/repeat_soft/std": 0.055666882544755936, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.4211280643939972, "rewards/total_composite/std": 0.1816686987876892, "reward": 0.4211280643939972, "reward_std": 0.1816686987876892, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14020542800426483, "sampling/sampling_logp_difference/max": 1.2822426557540894, "sampling/importance_sampling_ratio/min": 0.2774144411087036, "sampling/importance_sampling_ratio/mean": 1.0215743780136108, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9193695932626724, "clip_ratio/low_mean": 0.01923076994717121, "clip_ratio/low_min": 0.01923076994717121, "clip_ratio/high_mean": 0.09595720563083887, "clip_ratio/high_max": 0.09595720563083887, "clip_ratio/region_mean": 0.11518797557801008, "reward_total_mean": 0.4211280643939972, "reward_meter_mean": 0.5142291784286499, "reward_meter_std": 0.27461180090904236, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8774094581604004, "reward_repeat_soft_std": 0.055666882544755936, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.4211280643939972, "reward_total_composite_std": 0.1816686987876892} {"timestamp_utc": "2026-04-13T10:43:33Z", "mode": "train", "global_step": 1359, "epoch": 0.1365143144148669, "loss": 0.0196, "grad_norm": 12.106054306030273, "learning_rate": 5.884848484848486e-06, "num_tokens": 2399517.0, "completions/mean_length": 23.625, "completions/min_length": 20.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.625, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.8665356040000916, "rewards/meter/std": 0.3499932289123535, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9548467397689819, "rewards/repeat_soft/std": 0.020502854138612747, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.5870592594146729, "rewards/total_composite/std": 0.09871426224708557, "reward": 0.5870592594146729, "reward_std": 0.09871426224708557, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09746392071247101, "sampling/sampling_logp_difference/max": 1.053166389465332, "sampling/importance_sampling_ratio/min": 0.348831444978714, "sampling/importance_sampling_ratio/mean": 1.0364505052566528, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6199148334562778, "clip_ratio/low_mean": 0.010416666977107525, "clip_ratio/low_min": 0.010416666977107525, "clip_ratio/high_mean": 0.08211747603490949, "clip_ratio/high_max": 0.08211747603490949, "clip_ratio/region_mean": 0.09253414301201701, "reward_total_mean": 0.5870592594146729, "reward_meter_mean": 0.8665356040000916, "reward_meter_std": 0.3499932289123535, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9548467397689819, "reward_repeat_soft_std": 0.020502854138612747, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.5870592594146729, "reward_total_composite_std": 0.09871426224708557} {"timestamp_utc": "2026-04-13T10:43:39Z", "mode": "train", "global_step": 1360, "epoch": 0.1366147664490206, "loss": 0.1873, "grad_norm": 11.9026460647583, "learning_rate": 5.881818181818182e-06, "num_tokens": 2400959.0, "completions/mean_length": 30.25, "completions/min_length": 21.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.25, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.5686236023902893, "rewards/meter/std": 0.2765858471393585, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9673341512680054, "rewards/repeat_soft/std": 0.059099845588207245, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.4919586479663849, "rewards/total_composite/std": 0.09874510020017624, "reward": 0.4919586479663849, "reward_std": 0.09874510020017624, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11446832865476608, "sampling/sampling_logp_difference/max": 1.4261484146118164, "sampling/importance_sampling_ratio/min": 0.2402324229478836, "sampling/importance_sampling_ratio/mean": 1.0121992826461792, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4771619662642479, "clip_ratio/low_mean": 0.03084415663033724, "clip_ratio/low_min": 0.03084415663033724, "clip_ratio/high_mean": 0.06924877362325788, "clip_ratio/high_max": 0.06924877362325788, "clip_ratio/region_mean": 0.10009293025359511, "reward_total_mean": 0.4919586479663849, "reward_meter_mean": 0.5686236023902893, "reward_meter_std": 0.2765858471393585, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9673341512680054, "reward_repeat_soft_std": 0.059099845588207245, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.4919586479663849, "reward_total_composite_std": 0.09874510020017624} {"timestamp_utc": "2026-04-13T10:43:45Z", "mode": "train", "global_step": 1361, "epoch": 0.13671521848317428, "loss": 0.0322, "grad_norm": 10.216547966003418, "learning_rate": 5.878787878787879e-06, "num_tokens": 2402972.0, "completions/mean_length": 64.625, "completions/min_length": 51.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.625, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.29268279671669006, "rewards/meter/std": 0.2749910354614258, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.92520672082901, "rewards/repeat_soft/std": 0.04358210787177086, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.4158943295478821, "rewards/total_composite/std": 0.07025546580553055, "reward": 0.4158943295478821, "reward_std": 0.07025546580553055, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1270250529050827, "sampling/sampling_logp_difference/max": 1.696502685546875, "sampling/importance_sampling_ratio/min": 0.18332353234291077, "sampling/importance_sampling_ratio/mean": 1.0163600444793701, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6991370767354965, "clip_ratio/low_mean": 0.06225152686238289, "clip_ratio/low_min": 0.06225152686238289, "clip_ratio/high_mean": 0.04847604315727949, "clip_ratio/high_max": 0.04847604315727949, "clip_ratio/region_mean": 0.11072757001966238, "reward_total_mean": 0.4158943295478821, "reward_meter_mean": 0.29268279671669006, "reward_meter_std": 0.2749910354614258, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.92520672082901, "reward_repeat_soft_std": 0.04358210787177086, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.4158943295478821, "reward_total_composite_std": 0.07025546580553055} {"timestamp_utc": "2026-04-13T10:43:52Z", "mode": "train", "global_step": 1362, "epoch": 0.136815670517328, "loss": 0.2264, "grad_norm": 26.244260787963867, "learning_rate": 5.875757575757576e-06, "num_tokens": 2404385.0, "completions/mean_length": 25.625, "completions/min_length": 16.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.625, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.7774785161018372, "rewards/meter/std": 0.36153045296669006, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.5175492167472839, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9274853467941284, "rewards/repeat_soft/std": 0.06705766171216965, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.0645727664232254, "rewards/total_composite/mean": 0.4675617814064026, "rewards/total_composite/std": 0.13928283751010895, "reward": 0.4675617814064026, "reward_std": 0.13928283751010895, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16288518905639648, "sampling/sampling_logp_difference/max": 1.2759896516799927, "sampling/importance_sampling_ratio/min": 0.27915456891059875, "sampling/importance_sampling_ratio/mean": 1.0186101198196411, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9245611876249313, "clip_ratio/low_mean": 0.059489935636520386, "clip_ratio/low_min": 0.059489935636520386, "clip_ratio/high_mean": 0.055330883245915174, "clip_ratio/high_max": 0.055330883245915174, "clip_ratio/region_mean": 0.11482081888243556, "reward_total_mean": 0.4675617814064026, "reward_meter_mean": 0.7774785161018372, "reward_meter_std": 0.36153045296669006, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.5175492167472839, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9274853467941284, "reward_repeat_soft_std": 0.06705766171216965, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.0645727664232254, "reward_total_composite_mean": 0.4675617814064026, "reward_total_composite_std": 0.13928283751010895} {"timestamp_utc": "2026-04-13T10:43:59Z", "mode": "train", "global_step": 1363, "epoch": 0.13691612255148167, "loss": 0.0196, "grad_norm": 12.606419563293457, "learning_rate": 5.872727272727273e-06, "num_tokens": 2406050.0, "completions/mean_length": 43.125, "completions/min_length": 40.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.125, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.8502081632614136, "rewards/meter/std": 0.2545948028564453, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8542707562446594, "rewards/repeat_soft/std": 0.09567058831453323, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.5650582313537598, "rewards/total_composite/std": 0.07314696907997131, "reward": 0.5650582313537598, "reward_std": 0.07314696162939072, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14268463850021362, "sampling/sampling_logp_difference/max": 1.8414602279663086, "sampling/importance_sampling_ratio/min": 0.15858569741249084, "sampling/importance_sampling_ratio/mean": 1.026253342628479, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9797668755054474, "clip_ratio/low_mean": 0.05472383834421635, "clip_ratio/low_min": 0.05472383834421635, "clip_ratio/high_mean": 0.09119397960603237, "clip_ratio/high_max": 0.09119397960603237, "clip_ratio/region_mean": 0.14591781795024872, "reward_total_mean": 0.5650582313537598, "reward_meter_mean": 0.8502081632614136, "reward_meter_std": 0.2545948028564453, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8542707562446594, "reward_repeat_soft_std": 0.09567058831453323, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.5650582313537598, "reward_total_composite_std": 0.07314696907997131} {"timestamp_utc": "2026-04-13T10:44:05Z", "mode": "train", "global_step": 1364, "epoch": 0.13701657458563535, "loss": 0.029, "grad_norm": 10.385464668273926, "learning_rate": 5.8696969696969694e-06, "num_tokens": 2407574.0, "completions/mean_length": 38.5, "completions/min_length": 31.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.5, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.882033109664917, "rewards/meter/std": 0.30539917945861816, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7661975026130676, "rewards/repeat_soft/std": 0.1480618566274643, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.5629356503486633, "rewards/total_composite/std": 0.09463868290185928, "reward": 0.5629356503486633, "reward_std": 0.09463869035243988, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11570446193218231, "sampling/sampling_logp_difference/max": 1.1335511207580566, "sampling/importance_sampling_ratio/min": 0.32188817858695984, "sampling/importance_sampling_ratio/mean": 1.0171159505844116, "sampling/importance_sampling_ratio/max": 1.6529345512390137, "entropy": 0.9708805829286575, "clip_ratio/low_mean": 0.00872093066573143, "clip_ratio/low_min": 0.00872093066573143, "clip_ratio/high_mean": 0.11443972727283835, "clip_ratio/high_max": 0.11443972727283835, "clip_ratio/region_mean": 0.12316065793856978, "reward_total_mean": 0.5629356503486633, "reward_meter_mean": 0.882033109664917, "reward_meter_std": 0.30539917945861816, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7661975026130676, "reward_repeat_soft_std": 0.1480618566274643, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.5629356503486633, "reward_total_composite_std": 0.09463868290185928} {"timestamp_utc": "2026-04-13T10:44:11Z", "mode": "train", "global_step": 1365, "epoch": 0.13711702661978906, "loss": 0.0816, "grad_norm": 14.100876808166504, "learning_rate": 5.8666666666666675e-06, "num_tokens": 2409165.0, "completions/mean_length": 37.875, "completions/min_length": 34.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.875, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.47339460253715515, "rewards/meter/std": 0.37978094816207886, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9774109125137329, "rewards/repeat_soft/std": 0.028187068179249763, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.47740429639816284, "rewards/total_composite/std": 0.10387453436851501, "reward": 0.47740429639816284, "reward_std": 0.10387454181909561, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13374821841716766, "sampling/sampling_logp_difference/max": 1.9290554523468018, "sampling/importance_sampling_ratio/min": 0.14528536796569824, "sampling/importance_sampling_ratio/mean": 1.0020943880081177, "sampling/importance_sampling_ratio/max": 1.8436821699142456, "entropy": 0.8350476697087288, "clip_ratio/low_mean": 0.029022637056186795, "clip_ratio/low_min": 0.029022637056186795, "clip_ratio/high_mean": 0.08177459705621004, "clip_ratio/high_max": 0.08177459705621004, "clip_ratio/region_mean": 0.11079723411239684, "reward_total_mean": 0.47740429639816284, "reward_meter_mean": 0.47339460253715515, "reward_meter_std": 0.37978094816207886, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9774109125137329, "reward_repeat_soft_std": 0.028187068179249763, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.47740429639816284, "reward_total_composite_std": 0.10387453436851501} {"timestamp_utc": "2026-04-13T10:44:17Z", "mode": "train", "global_step": 1366, "epoch": 0.13721747865394274, "loss": 0.0075, "grad_norm": 15.947189331054688, "learning_rate": 5.863636363636364e-06, "num_tokens": 2410658.0, "completions/mean_length": 38.625, "completions/min_length": 36.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.625, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.9263715744018555, "rewards/meter/std": 0.0999070331454277, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8989310264587402, "rewards/repeat_soft/std": 0.025368811562657356, "rewards/judge_quality/mean": 0.5387499928474426, "rewards/judge_quality/std": 0.18192915618419647, "rewards/total_composite/mean": 0.657719075679779, "rewards/total_composite/std": 0.10532543063163757, "reward": 0.657719075679779, "reward_std": 0.10532545298337936, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12557724118232727, "sampling/sampling_logp_difference/max": 1.2217206954956055, "sampling/importance_sampling_ratio/min": 0.2947225868701935, "sampling/importance_sampling_ratio/mean": 1.012650966644287, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8277973011136055, "clip_ratio/low_mean": 0.0658501754514873, "clip_ratio/low_min": 0.0658501754514873, "clip_ratio/high_mean": 0.022368420846760273, "clip_ratio/high_max": 0.022368420846760273, "clip_ratio/region_mean": 0.08821859629824758, "reward_total_mean": 0.657719075679779, "reward_meter_mean": 0.9263715744018555, "reward_meter_std": 0.0999070331454277, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8989310264587402, "reward_repeat_soft_std": 0.025368811562657356, "reward_judge_quality_mean": 0.5387499928474426, "reward_judge_quality_std": 0.18192915618419647, "reward_total_composite_mean": 0.657719075679779, "reward_total_composite_std": 0.10532543063163757} {"timestamp_utc": "2026-04-13T10:44:23Z", "mode": "train", "global_step": 1367, "epoch": 0.13731793068809645, "loss": 0.0601, "grad_norm": 8.43343734741211, "learning_rate": 5.860606060606061e-06, "num_tokens": 2412634.0, "completions/mean_length": 76.0, "completions/min_length": 67.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.9826860427856445, "rewards/meter/std": 0.006795603781938553, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7383664846420288, "rewards/repeat_soft/std": 0.07205822318792343, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.6053905487060547, "rewards/total_composite/std": 0.12587036192417145, "reward": 0.6053905487060547, "reward_std": 0.12587036192417145, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09169119596481323, "sampling/sampling_logp_difference/max": 1.7090654373168945, "sampling/importance_sampling_ratio/min": 0.1810348927974701, "sampling/importance_sampling_ratio/mean": 1.0142838954925537, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6261106580495834, "clip_ratio/low_mean": 0.07950532995164394, "clip_ratio/low_min": 0.07950532995164394, "clip_ratio/high_mean": 0.01689189113676548, "clip_ratio/high_max": 0.01689189113676548, "clip_ratio/region_mean": 0.09639722108840942, "reward_total_mean": 0.6053905487060547, "reward_meter_mean": 0.9826860427856445, "reward_meter_std": 0.006795603781938553, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7383664846420288, "reward_repeat_soft_std": 0.07205822318792343, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.6053905487060547, "reward_total_composite_std": 0.12587036192417145} {"timestamp_utc": "2026-04-13T10:44:30Z", "mode": "train", "global_step": 1368, "epoch": 0.13741838272225013, "loss": -0.0401, "grad_norm": 16.965978622436523, "learning_rate": 5.8575757575757584e-06, "num_tokens": 2414299.0, "completions/mean_length": 36.125, "completions/min_length": 30.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.125, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.8173834085464478, "rewards/meter/std": 0.2950228452682495, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9791263341903687, "rewards/repeat_soft/std": 0.032146155834198, "rewards/judge_quality/mean": 0.5275000333786011, "rewards/judge_quality/std": 0.1954299360513687, "rewards/total_composite/mean": 0.5984628796577454, "rewards/total_composite/std": 0.09653806686401367, "reward": 0.5984628796577454, "reward_std": 0.09653805196285248, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14996446669101715, "sampling/sampling_logp_difference/max": 1.7358055114746094, "sampling/importance_sampling_ratio/min": 0.1762581765651703, "sampling/importance_sampling_ratio/mean": 1.0238789319992065, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8199257031083107, "clip_ratio/low_mean": 0.052777779288589954, "clip_ratio/low_min": 0.052777779288589954, "clip_ratio/high_mean": 0.09061235003173351, "clip_ratio/high_max": 0.09061235003173351, "clip_ratio/region_mean": 0.14339012932032347, "reward_total_mean": 0.5984628796577454, "reward_meter_mean": 0.8173834085464478, "reward_meter_std": 0.2950228452682495, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9791263341903687, "reward_repeat_soft_std": 0.032146155834198, "reward_judge_quality_mean": 0.5275000333786011, "reward_judge_quality_std": 0.1954299360513687, "reward_total_composite_mean": 0.5984628796577454, "reward_total_composite_std": 0.09653806686401367} {"timestamp_utc": "2026-04-13T10:44:42Z", "mode": "train", "global_step": 1369, "epoch": 0.1375188347564038, "loss": -0.138, "grad_norm": 2.002079963684082, "learning_rate": 5.854545454545455e-06, "num_tokens": 2416019.0, "completions/mean_length": 112.0, "completions/min_length": 46.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 54.857147216796875, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.7430543899536133, "rewards/meter/std": 0.28496313095092773, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.7456576824188232, "rewards/repeat_soft/std": 0.07440777122974396, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.13274572789669037, "rewards/total_composite/mean": 0.46270573139190674, "rewards/total_composite/std": 0.19669091701507568, "reward": 0.46270573139190674, "reward_std": 0.19669093191623688, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09795010089874268, "sampling/sampling_logp_difference/max": 1.4368808269500732, "sampling/importance_sampling_ratio/min": 0.23766793310642242, "sampling/importance_sampling_ratio/mean": 1.0065462589263916, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4737773537635803, "clip_ratio/low_mean": 0.010775862261652946, "clip_ratio/low_min": 0.010775862261652946, "clip_ratio/high_mean": 0.06216462189331651, "clip_ratio/high_max": 0.06216462189331651, "clip_ratio/region_mean": 0.07294048415496945, "reward_total_mean": 0.46270573139190674, "reward_meter_mean": 0.7430543899536133, "reward_meter_std": 0.28496313095092773, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.7456576824188232, "reward_repeat_soft_std": 0.07440777122974396, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.13274572789669037, "reward_total_composite_mean": 0.46270573139190674, "reward_total_composite_std": 0.19669091701507568} {"timestamp_utc": "2026-04-13T10:44:49Z", "mode": "train", "global_step": 1370, "epoch": 0.13761928679055752, "loss": 0.0231, "grad_norm": 10.684398651123047, "learning_rate": 5.851515151515152e-06, "num_tokens": 2417878.0, "completions/mean_length": 67.375, "completions/min_length": 52.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.375, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.8285349607467651, "rewards/meter/std": 0.17354312539100647, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8799509406089783, "rewards/repeat_soft/std": 0.07719051092863083, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.5655603408813477, "rewards/total_composite/std": 0.10099063813686371, "reward": 0.5655603408813477, "reward_std": 0.10099063068628311, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12360924482345581, "sampling/sampling_logp_difference/max": 2.207282543182373, "sampling/importance_sampling_ratio/min": 0.1099991574883461, "sampling/importance_sampling_ratio/mean": 1.0059648752212524, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.764433428645134, "clip_ratio/low_mean": 0.04938335530459881, "clip_ratio/low_min": 0.04938335530459881, "clip_ratio/high_mean": 0.05147484363988042, "clip_ratio/high_max": 0.05147484363988042, "clip_ratio/region_mean": 0.10085819894447923, "reward_total_mean": 0.5655603408813477, "reward_meter_mean": 0.8285349607467651, "reward_meter_std": 0.17354312539100647, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8799509406089783, "reward_repeat_soft_std": 0.07719051092863083, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.5655603408813477, "reward_total_composite_std": 0.10099063813686371} {"timestamp_utc": "2026-04-13T10:44:54Z", "mode": "train", "global_step": 1371, "epoch": 0.1377197388247112, "loss": -0.0581, "grad_norm": 15.889586448669434, "learning_rate": 5.8484848484848485e-06, "num_tokens": 2419231.0, "completions/mean_length": 20.125, "completions/min_length": 16.0, "completions/max_length": 25.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 20.125, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 25.0, "rewards/meter/mean": 0.9839861989021301, "rewards/meter/std": 0.017121868208050728, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9169131517410278, "rewards/repeat_soft/std": 0.048553213477134705, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.0843462198972702, "rewards/total_composite/mean": 0.5842519998550415, "rewards/total_composite/std": 0.05528772622346878, "reward": 0.5842519998550415, "reward_std": 0.05528773367404938, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14024120569229126, "sampling/sampling_logp_difference/max": 1.1625232696533203, "sampling/importance_sampling_ratio/min": 0.3126961886882782, "sampling/importance_sampling_ratio/mean": 1.0222989320755005, "sampling/importance_sampling_ratio/max": 1.7781554460525513, "entropy": 1.002502478659153, "clip_ratio/low_mean": 0.022518382407724857, "clip_ratio/low_min": 0.022518382407724857, "clip_ratio/high_mean": 0.12860644469037652, "clip_ratio/high_max": 0.12860644469037652, "clip_ratio/region_mean": 0.15112482709810138, "reward_total_mean": 0.5842519998550415, "reward_meter_mean": 0.9839861989021301, "reward_meter_std": 0.017121868208050728, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9169131517410278, "reward_repeat_soft_std": 0.048553213477134705, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.0843462198972702, "reward_total_composite_mean": 0.5842519998550415, "reward_total_composite_std": 0.05528772622346878} {"timestamp_utc": "2026-04-13T10:45:00Z", "mode": "train", "global_step": 1372, "epoch": 0.1378201908588649, "loss": 0.0497, "grad_norm": 11.172639846801758, "learning_rate": 5.845454545454547e-06, "num_tokens": 2420871.0, "completions/mean_length": 32.0, "completions/min_length": 29.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9418027400970459, "rewards/meter/std": 0.08665227890014648, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7936781644821167, "rewards/repeat_soft/std": 0.04992871358990669, "rewards/judge_quality/mean": 0.7950000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.8017357587814331, "rewards/total_composite/std": 0.14127589762210846, "reward": 0.8017357587814331, "reward_std": 0.14127591252326965, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08830776065587997, "sampling/sampling_logp_difference/max": 1.2683124542236328, "sampling/importance_sampling_ratio/min": 0.2813059389591217, "sampling/importance_sampling_ratio/mean": 1.003429889678955, "sampling/importance_sampling_ratio/max": 1.9754638671875, "entropy": 0.5145772732794285, "clip_ratio/low_mean": 0.042360677383840084, "clip_ratio/low_min": 0.042360677383840084, "clip_ratio/high_mean": 0.023839198518544436, "clip_ratio/high_max": 0.023839198518544436, "clip_ratio/region_mean": 0.06619987590238452, "reward_total_mean": 0.8017357587814331, "reward_meter_mean": 0.9418027400970459, "reward_meter_std": 0.08665227890014648, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7936781644821167, "reward_repeat_soft_std": 0.04992871358990669, "reward_judge_quality_mean": 0.7950000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.8017357587814331, "reward_total_composite_std": 0.14127589762210846} {"timestamp_utc": "2026-04-13T10:45:07Z", "mode": "train", "global_step": 1373, "epoch": 0.13792064289301859, "loss": 0.0186, "grad_norm": 9.985396385192871, "learning_rate": 5.842424242424243e-06, "num_tokens": 2423047.0, "completions/mean_length": 95.0, "completions/min_length": 86.0, "completions/max_length": 104.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.0, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.5731029510498047, "rewards/meter/std": 0.37828680872917175, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.838218092918396, "rewards/repeat_soft/std": 0.07607004791498184, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.482189804315567, "rewards/total_composite/std": 0.09745136648416519, "reward": 0.482189804315567, "reward_std": 0.0974513590335846, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0944228395819664, "sampling/sampling_logp_difference/max": 1.4184298515319824, "sampling/importance_sampling_ratio/min": 0.24209386110305786, "sampling/importance_sampling_ratio/mean": 1.0090051889419556, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.587487243115902, "clip_ratio/low_mean": 0.036595333367586136, "clip_ratio/low_min": 0.036595333367586136, "clip_ratio/high_mean": 0.050420207902789116, "clip_ratio/high_max": 0.050420207902789116, "clip_ratio/region_mean": 0.08701554127037525, "reward_total_mean": 0.482189804315567, "reward_meter_mean": 0.5731029510498047, "reward_meter_std": 0.37828680872917175, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.838218092918396, "reward_repeat_soft_std": 0.07607004791498184, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.482189804315567, "reward_total_composite_std": 0.09745136648416519} {"timestamp_utc": "2026-04-13T10:45:14Z", "mode": "train", "global_step": 1374, "epoch": 0.13802109492717227, "loss": 0.0507, "grad_norm": 12.163618087768555, "learning_rate": 5.83939393939394e-06, "num_tokens": 2425055.0, "completions/mean_length": 79.0, "completions/min_length": 70.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.0, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.5774952173233032, "rewards/meter/std": 0.26382485032081604, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8827155232429504, "rewards/repeat_soft/std": 0.06778127700090408, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.49006354808807373, "rewards/total_composite/std": 0.06766549497842789, "reward": 0.49006354808807373, "reward_std": 0.06766548752784729, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11644869297742844, "sampling/sampling_logp_difference/max": 2.140930652618408, "sampling/importance_sampling_ratio/min": 0.11754540354013443, "sampling/importance_sampling_ratio/mean": 0.9974746704101562, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5163152329623699, "clip_ratio/low_mean": 0.0407986119389534, "clip_ratio/low_min": 0.0407986119389534, "clip_ratio/high_mean": 0.07150247041136026, "clip_ratio/high_max": 0.07150247041136026, "clip_ratio/region_mean": 0.11230108235031366, "reward_total_mean": 0.49006354808807373, "reward_meter_mean": 0.5774952173233032, "reward_meter_std": 0.26382485032081604, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8827155232429504, "reward_repeat_soft_std": 0.06778127700090408, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.49006354808807373, "reward_total_composite_std": 0.06766549497842789} {"timestamp_utc": "2026-04-13T10:45:21Z", "mode": "train", "global_step": 1375, "epoch": 0.13812154696132597, "loss": 0.1259, "grad_norm": 9.925308227539062, "learning_rate": 5.836363636363637e-06, "num_tokens": 2427042.0, "completions/mean_length": 85.375, "completions/min_length": 68.0, "completions/max_length": 111.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.375, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.5373973846435547, "rewards/meter/std": 0.3738399147987366, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8409735560417175, "rewards/repeat_soft/std": 0.06459219008684158, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.4707145690917969, "rewards/total_composite/std": 0.10587940365076065, "reward": 0.4707145690917969, "reward_std": 0.10587940365076065, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10384798049926758, "sampling/sampling_logp_difference/max": 2.5874240398406982, "sampling/importance_sampling_ratio/min": 0.07521353662014008, "sampling/importance_sampling_ratio/mean": 1.0010960102081299, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5792411118745804, "clip_ratio/low_mean": 0.04503625771030784, "clip_ratio/low_min": 0.04503625771030784, "clip_ratio/high_mean": 0.05577491782605648, "clip_ratio/high_max": 0.05577491782605648, "clip_ratio/region_mean": 0.10081117553636432, "reward_total_mean": 0.4707145690917969, "reward_meter_mean": 0.5373973846435547, "reward_meter_std": 0.3738399147987366, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8409735560417175, "reward_repeat_soft_std": 0.06459219008684158, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.4707145690917969, "reward_total_composite_std": 0.10587940365076065} {"timestamp_utc": "2026-04-13T10:45:32Z", "mode": "train", "global_step": 1376, "epoch": 0.13822199899547966, "loss": -0.1087, "grad_norm": 1.4767781496047974, "learning_rate": 5.833333333333334e-06, "num_tokens": 2428574.0, "completions/mean_length": 93.5, "completions/min_length": 31.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 33.71428680419922, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.8505613803863525, "rewards/meter/std": 0.326339989900589, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8759030103683472, "rewards/repeat_soft/std": 0.07500092685222626, "rewards/judge_quality/mean": 0.39249998331069946, "rewards/judge_quality/std": 0.13905291259288788, "rewards/total_composite/mean": 0.530614972114563, "rewards/total_composite/std": 0.215074360370636, "reward": 0.530614972114563, "reward_std": 0.21507437527179718, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12482795119285583, "sampling/sampling_logp_difference/max": 1.912602186203003, "sampling/importance_sampling_ratio/min": 0.14769555628299713, "sampling/importance_sampling_ratio/mean": 0.9998764991760254, "sampling/importance_sampling_ratio/max": 1.6154453754425049, "entropy": 0.688353419303894, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.10693441238254309, "clip_ratio/high_max": 0.10693441238254309, "clip_ratio/region_mean": 0.10693441238254309, "reward_total_mean": 0.530614972114563, "reward_meter_mean": 0.8505613803863525, "reward_meter_std": 0.326339989900589, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8759030103683472, "reward_repeat_soft_std": 0.07500092685222626, "reward_judge_quality_mean": 0.39249998331069946, "reward_judge_quality_std": 0.13905291259288788, "reward_total_composite_mean": 0.530614972114563, "reward_total_composite_std": 0.215074360370636} {"timestamp_utc": "2026-04-13T10:45:43Z", "mode": "train", "global_step": 1377, "epoch": 0.13832245102963334, "loss": -0.1533, "grad_norm": 1.7564753293991089, "learning_rate": 5.83030303030303e-06, "num_tokens": 2430299.0, "completions/mean_length": 115.625, "completions/min_length": 53.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 59.000003814697266, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.8179754018783569, "rewards/meter/std": 0.17808018624782562, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.6838929653167725, "rewards/repeat_soft/std": 0.06658895313739777, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.4737984240055084, "rewards/total_composite/std": 0.19232341647148132, "reward": 0.4737984240055084, "reward_std": 0.19232341647148132, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09157626330852509, "sampling/sampling_logp_difference/max": 2.145017623901367, "sampling/importance_sampling_ratio/min": 0.11706597357988358, "sampling/importance_sampling_ratio/mean": 1.0068401098251343, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4960397928953171, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.08159125922247767, "clip_ratio/high_max": 0.08159125922247767, "clip_ratio/region_mean": 0.08159125922247767, "reward_total_mean": 0.4737984240055084, "reward_meter_mean": 0.8179754018783569, "reward_meter_std": 0.17808018624782562, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.6838929653167725, "reward_repeat_soft_std": 0.06658895313739777, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.4737984240055084, "reward_total_composite_std": 0.19232341647148132} {"timestamp_utc": "2026-04-13T10:45:55Z", "mode": "train", "global_step": 1378, "epoch": 0.13842290306378705, "loss": -0.1543, "grad_norm": 2.857304334640503, "learning_rate": 5.8272727272727285e-06, "num_tokens": 2432266.0, "completions/mean_length": 127.875, "completions/min_length": 69.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 73.0, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.6391924619674683, "rewards/meter/std": 0.36736181378364563, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.80765700340271, "rewards/repeat_soft/std": 0.09302859008312225, "rewards/judge_quality/mean": 0.38999998569488525, "rewards/judge_quality/std": 0.18845234811306, "rewards/total_composite/mean": 0.4120762348175049, "rewards/total_composite/std": 0.18186332285404205, "reward": 0.4120762348175049, "reward_std": 0.18186332285404205, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13007330894470215, "sampling/sampling_logp_difference/max": 1.6582809686660767, "sampling/importance_sampling_ratio/min": 0.19046610593795776, "sampling/importance_sampling_ratio/mean": 1.0254606008529663, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6915091574192047, "clip_ratio/low_mean": 0.025597664527595043, "clip_ratio/low_min": 0.025597664527595043, "clip_ratio/high_mean": 0.0639688279479742, "clip_ratio/high_max": 0.0639688279479742, "clip_ratio/region_mean": 0.08956649247556925, "reward_total_mean": 0.4120762348175049, "reward_meter_mean": 0.6391924619674683, "reward_meter_std": 0.36736181378364563, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.80765700340271, "reward_repeat_soft_std": 0.09302859008312225, "reward_judge_quality_mean": 0.38999998569488525, "reward_judge_quality_std": 0.18845234811306, "reward_total_composite_mean": 0.4120762348175049, "reward_total_composite_std": 0.18186332285404205} {"timestamp_utc": "2026-04-13T10:46:01Z", "mode": "train", "global_step": 1379, "epoch": 0.13852335509794073, "loss": 0.0187, "grad_norm": 14.179174423217773, "learning_rate": 5.824242424242425e-06, "num_tokens": 2433671.0, "completions/mean_length": 19.625, "completions/min_length": 18.0, "completions/max_length": 21.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 19.625, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 21.0, "rewards/meter/mean": 0.980178952217102, "rewards/meter/std": 0.017566950991749763, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8862823247909546, "rewards/repeat_soft/std": 0.06156831234693527, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.6125080585479736, "rewards/total_composite/std": 0.016003286466002464, "reward": 0.6125080585479736, "reward_std": 0.016003292053937912, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08959906548261642, "sampling/sampling_logp_difference/max": 0.9062271118164062, "sampling/importance_sampling_ratio/min": 0.4040457606315613, "sampling/importance_sampling_ratio/mean": 1.0124434232711792, "sampling/importance_sampling_ratio/max": 1.6209818124771118, "entropy": 0.6474045962095261, "clip_ratio/low_mean": 0.03131265705451369, "clip_ratio/low_min": 0.03131265705451369, "clip_ratio/high_mean": 0.052736007142812014, "clip_ratio/high_max": 0.052736007142812014, "clip_ratio/region_mean": 0.0840486641973257, "reward_total_mean": 0.6125080585479736, "reward_meter_mean": 0.980178952217102, "reward_meter_std": 0.017566950991749763, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8862823247909546, "reward_repeat_soft_std": 0.06156831234693527, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.6125080585479736, "reward_total_composite_std": 0.016003286466002464} {"timestamp_utc": "2026-04-13T10:46:07Z", "mode": "train", "global_step": 1380, "epoch": 0.13862380713209443, "loss": -0.0135, "grad_norm": 6.226532936096191, "learning_rate": 5.821212121212122e-06, "num_tokens": 2435696.0, "completions/mean_length": 79.125, "completions/min_length": 71.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.125, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.9815012812614441, "rewards/meter/std": 0.021678613498806953, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7070019841194153, "rewards/repeat_soft/std": 0.11901603639125824, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.546882152557373, "rewards/total_composite/std": 0.06000892072916031, "reward": 0.546882152557373, "reward_std": 0.06000891700387001, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10609939694404602, "sampling/sampling_logp_difference/max": 1.5791711807250977, "sampling/importance_sampling_ratio/min": 0.20614588260650635, "sampling/importance_sampling_ratio/mean": 1.0092766284942627, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6289454698562622, "clip_ratio/low_mean": 0.017489806516095996, "clip_ratio/low_min": 0.017489806516095996, "clip_ratio/high_mean": 0.07526215631514788, "clip_ratio/high_max": 0.07526215631514788, "clip_ratio/region_mean": 0.09275196283124387, "reward_total_mean": 0.546882152557373, "reward_meter_mean": 0.9815012812614441, "reward_meter_std": 0.021678613498806953, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7070019841194153, "reward_repeat_soft_std": 0.11901603639125824, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.546882152557373, "reward_total_composite_std": 0.06000892072916031} {"timestamp_utc": "2026-04-13T10:46:14Z", "mode": "train", "global_step": 1381, "epoch": 0.13872425916624812, "loss": 0.0016, "grad_norm": 9.563642501831055, "learning_rate": 5.8181818181818185e-06, "num_tokens": 2437704.0, "completions/mean_length": 72.0, "completions/min_length": 64.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.9264421463012695, "rewards/meter/std": 0.095173180103302, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8494632244110107, "rewards/repeat_soft/std": 0.08679043501615524, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5803382396697998, "rewards/total_composite/std": 0.024983327835798264, "reward": 0.5803382396697998, "reward_std": 0.024983324110507965, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11786331236362457, "sampling/sampling_logp_difference/max": 1.7849845886230469, "sampling/importance_sampling_ratio/min": 0.16779965162277222, "sampling/importance_sampling_ratio/mean": 1.0110652446746826, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7982779815793037, "clip_ratio/low_mean": 0.05310388561338186, "clip_ratio/low_min": 0.05310388561338186, "clip_ratio/high_mean": 0.06156429462134838, "clip_ratio/high_max": 0.06156429462134838, "clip_ratio/region_mean": 0.11466818023473024, "reward_total_mean": 0.5803382396697998, "reward_meter_mean": 0.9264421463012695, "reward_meter_std": 0.095173180103302, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8494632244110107, "reward_repeat_soft_std": 0.08679043501615524, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5803382396697998, "reward_total_composite_std": 0.024983327835798264} {"timestamp_utc": "2026-04-13T10:46:20Z", "mode": "train", "global_step": 1382, "epoch": 0.1388247112004018, "loss": -0.0187, "grad_norm": 7.579502582550049, "learning_rate": 5.815151515151516e-06, "num_tokens": 2439593.0, "completions/mean_length": 59.125, "completions/min_length": 54.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.125, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.8978123664855957, "rewards/meter/std": 0.1514429748058319, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6922234296798706, "rewards/repeat_soft/std": 0.048303522169589996, "rewards/judge_quality/mean": 0.41874998807907104, "rewards/judge_quality/std": 0.21931305527687073, "rewards/total_composite/mean": 0.5479089021682739, "rewards/total_composite/std": 0.1442357897758484, "reward": 0.5479089021682739, "reward_std": 0.1442357748746872, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08081500232219696, "sampling/sampling_logp_difference/max": 1.7496185302734375, "sampling/importance_sampling_ratio/min": 0.1738402396440506, "sampling/importance_sampling_ratio/mean": 0.9880836606025696, "sampling/importance_sampling_ratio/max": 1.9317129850387573, "entropy": 0.3957347720861435, "clip_ratio/low_mean": 0.04596590343862772, "clip_ratio/low_min": 0.04596590343862772, "clip_ratio/high_mean": 0.037464292254298925, "clip_ratio/high_max": 0.037464292254298925, "clip_ratio/region_mean": 0.08343019569292665, "reward_total_mean": 0.5479089021682739, "reward_meter_mean": 0.8978123664855957, "reward_meter_std": 0.1514429748058319, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6922234296798706, "reward_repeat_soft_std": 0.048303522169589996, "reward_judge_quality_mean": 0.41874998807907104, "reward_judge_quality_std": 0.21931305527687073, "reward_total_composite_mean": 0.5479089021682739, "reward_total_composite_std": 0.1442357897758484} {"timestamp_utc": "2026-04-13T10:46:27Z", "mode": "train", "global_step": 1383, "epoch": 0.1389251632345555, "loss": 0.01, "grad_norm": 7.723628520965576, "learning_rate": 5.812121212121212e-06, "num_tokens": 2441537.0, "completions/mean_length": 70.0, "completions/min_length": 66.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.8299963474273682, "rewards/meter/std": 0.13433019816875458, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7724564075469971, "rewards/repeat_soft/std": 0.028093140572309494, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.531490683555603, "rewards/total_composite/std": 0.04767953231930733, "reward": 0.531490683555603, "reward_std": 0.04767955094575882, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0866677388548851, "sampling/sampling_logp_difference/max": 2.966364860534668, "sampling/importance_sampling_ratio/min": 0.05149014666676521, "sampling/importance_sampling_ratio/mean": 1.0031169652938843, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.42300498485565186, "clip_ratio/low_mean": 0.02848917990922928, "clip_ratio/low_min": 0.02848917990922928, "clip_ratio/high_mean": 0.02885281457565725, "clip_ratio/high_max": 0.02885281457565725, "clip_ratio/region_mean": 0.05734199448488653, "reward_total_mean": 0.531490683555603, "reward_meter_mean": 0.8299963474273682, "reward_meter_std": 0.13433019816875458, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7724564075469971, "reward_repeat_soft_std": 0.028093140572309494, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.531490683555603, "reward_total_composite_std": 0.04767953231930733} {"timestamp_utc": "2026-04-13T10:46:40Z", "mode": "train", "global_step": 1384, "epoch": 0.13902561526870919, "loss": -0.1201, "grad_norm": 2.4333841800689697, "learning_rate": 5.8090909090909095e-06, "num_tokens": 2443120.0, "completions/mean_length": 106.875, "completions/min_length": 44.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 49.000003814697266, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.26074719429016113, "rewards/meter/std": 0.245651975274086, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.793143630027771, "rewards/repeat_soft/std": 0.07925429940223694, "rewards/judge_quality/mean": 0.33124998211860657, "rewards/judge_quality/std": 0.13715866208076477, "rewards/total_composite/mean": 0.3378831744194031, "rewards/total_composite/std": 0.1523023545742035, "reward": 0.3378831744194031, "reward_std": 0.1523023545742035, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10070676356554031, "sampling/sampling_logp_difference/max": 2.923140525817871, "sampling/importance_sampling_ratio/min": 0.053764570504426956, "sampling/importance_sampling_ratio/mean": 1.0106135606765747, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.36102766171097755, "clip_ratio/low_mean": 0.014150943607091904, "clip_ratio/low_min": 0.014150943607091904, "clip_ratio/high_mean": 0.06851313123479486, "clip_ratio/high_max": 0.06851313123479486, "clip_ratio/region_mean": 0.08266407484188676, "reward_total_mean": 0.3378831744194031, "reward_meter_mean": 0.26074719429016113, "reward_meter_std": 0.245651975274086, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.793143630027771, "reward_repeat_soft_std": 0.07925429940223694, "reward_judge_quality_mean": 0.33124998211860657, "reward_judge_quality_std": 0.13715866208076477, "reward_total_composite_mean": 0.3378831744194031, "reward_total_composite_std": 0.1523023545742035} {"timestamp_utc": "2026-04-13T10:46:46Z", "mode": "train", "global_step": 1385, "epoch": 0.1391260673028629, "loss": -0.0587, "grad_norm": 15.846782684326172, "learning_rate": 5.806060606060606e-06, "num_tokens": 2444513.0, "completions/mean_length": 27.125, "completions/min_length": 23.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.125, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9157686233520508, "rewards/meter/std": 0.10688792914152145, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.713201642036438, "rewards/repeat_soft/std": 0.07831180840730667, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.5479727983474731, "rewards/total_composite/std": 0.06456225365400314, "reward": 0.5479727983474731, "reward_std": 0.06456225365400314, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09401027858257294, "sampling/sampling_logp_difference/max": 1.2109918594360352, "sampling/importance_sampling_ratio/min": 0.2979016602039337, "sampling/importance_sampling_ratio/mean": 0.9902519583702087, "sampling/importance_sampling_ratio/max": 1.700728416442871, "entropy": 0.46405426785349846, "clip_ratio/low_mean": 0.005434782709926367, "clip_ratio/low_min": 0.005434782709926367, "clip_ratio/high_mean": 0.049688469618558884, "clip_ratio/high_max": 0.049688469618558884, "clip_ratio/region_mean": 0.05512325232848525, "reward_total_mean": 0.5479727983474731, "reward_meter_mean": 0.9157686233520508, "reward_meter_std": 0.10688792914152145, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.713201642036438, "reward_repeat_soft_std": 0.07831180840730667, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.5479727983474731, "reward_total_composite_std": 0.06456225365400314} {"timestamp_utc": "2026-04-13T10:46:53Z", "mode": "train", "global_step": 1386, "epoch": 0.13922651933701657, "loss": 0.0188, "grad_norm": 10.31757640838623, "learning_rate": 5.803030303030304e-06, "num_tokens": 2446218.0, "completions/mean_length": 53.125, "completions/min_length": 45.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.125, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.8301163911819458, "rewards/meter/std": 0.29177701473236084, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7438743710517883, "rewards/repeat_soft/std": 0.0749485120177269, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.21357084810733795, "rewards/total_composite/mean": 0.5660555958747864, "rewards/total_composite/std": 0.14403384923934937, "reward": 0.5660555958747864, "reward_std": 0.14403383433818817, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0759248211979866, "sampling/sampling_logp_difference/max": 1.408792495727539, "sampling/importance_sampling_ratio/min": 0.2444382756948471, "sampling/importance_sampling_ratio/mean": 1.0056344270706177, "sampling/importance_sampling_ratio/max": 1.8468225002288818, "entropy": 0.4119080901145935, "clip_ratio/low_mean": 0.009722222341224551, "clip_ratio/low_min": 0.009722222341224551, "clip_ratio/high_mean": 0.04207471385598183, "clip_ratio/high_max": 0.04207471385598183, "clip_ratio/region_mean": 0.05179693619720638, "reward_total_mean": 0.5660555958747864, "reward_meter_mean": 0.8301163911819458, "reward_meter_std": 0.29177701473236084, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7438743710517883, "reward_repeat_soft_std": 0.0749485120177269, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.21357084810733795, "reward_total_composite_mean": 0.5660555958747864, "reward_total_composite_std": 0.14403384923934937} {"timestamp_utc": "2026-04-13T10:47:04Z", "mode": "train", "global_step": 1387, "epoch": 0.13932697137117026, "loss": -0.1763, "grad_norm": 1.929757833480835, "learning_rate": 5.8e-06, "num_tokens": 2448452.0, "completions/mean_length": 140.25, "completions/min_length": 79.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 87.14286041259766, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.9144402146339417, "rewards/meter/std": 0.1888921856880188, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8584533333778381, "rewards/repeat_soft/std": 0.05335886776447296, "rewards/judge_quality/mean": 0.4112499952316284, "rewards/judge_quality/std": 0.1797965168952942, "rewards/total_composite/mean": 0.5105130672454834, "rewards/total_composite/std": 0.21891987323760986, "reward": 0.5105130672454834, "reward_std": 0.21891987323760986, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10738111287355423, "sampling/sampling_logp_difference/max": 1.695192813873291, "sampling/importance_sampling_ratio/min": 0.18356384336948395, "sampling/importance_sampling_ratio/mean": 1.0207411050796509, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5428299009799957, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0956077417358756, "clip_ratio/high_max": 0.0956077417358756, "clip_ratio/region_mean": 0.0956077417358756, "reward_total_mean": 0.5105130672454834, "reward_meter_mean": 0.9144402146339417, "reward_meter_std": 0.1888921856880188, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8584533333778381, "reward_repeat_soft_std": 0.05335886776447296, "reward_judge_quality_mean": 0.4112499952316284, "reward_judge_quality_std": 0.1797965168952942, "reward_total_composite_mean": 0.5105130672454834, "reward_total_composite_std": 0.21891987323760986} {"timestamp_utc": "2026-04-13T10:47:11Z", "mode": "train", "global_step": 1388, "epoch": 0.13942742340532396, "loss": -0.0345, "grad_norm": 9.070403099060059, "learning_rate": 5.796969696969698e-06, "num_tokens": 2450718.0, "completions/mean_length": 93.25, "completions/min_length": 82.0, "completions/max_length": 106.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.25, "completions/min_terminated_length": 82.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.5084026455879211, "rewards/meter/std": 0.409898042678833, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7606096863746643, "rewards/repeat_soft/std": 0.05007196590304375, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.4384540021419525, "rewards/total_composite/std": 0.1053735613822937, "reward": 0.4384540021419525, "reward_std": 0.1053735539317131, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10059534758329391, "sampling/sampling_logp_difference/max": 1.3359971046447754, "sampling/importance_sampling_ratio/min": 0.2628959119319916, "sampling/importance_sampling_ratio/mean": 1.0168665647506714, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5884588398039341, "clip_ratio/low_mean": 0.038291629403829575, "clip_ratio/low_min": 0.038291629403829575, "clip_ratio/high_mean": 0.046265526209026575, "clip_ratio/high_max": 0.046265526209026575, "clip_ratio/region_mean": 0.08455715561285615, "reward_total_mean": 0.4384540021419525, "reward_meter_mean": 0.5084026455879211, "reward_meter_std": 0.409898042678833, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7606096863746643, "reward_repeat_soft_std": 0.05007196590304375, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.4384540021419525, "reward_total_composite_std": 0.1053735613822937} {"timestamp_utc": "2026-04-13T10:47:22Z", "mode": "train", "global_step": 1389, "epoch": 0.13952787543947764, "loss": -0.1142, "grad_norm": 2.0049874782562256, "learning_rate": 5.793939393939394e-06, "num_tokens": 2452175.0, "completions/mean_length": 96.125, "completions/min_length": 34.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 36.71428680419922, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.6961934566497803, "rewards/meter/std": 0.27985823154449463, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.817654013633728, "rewards/repeat_soft/std": 0.04878468066453934, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.13274572789669037, "rewards/total_composite/mean": 0.4702560305595398, "rewards/total_composite/std": 0.19222553074359894, "reward": 0.4702560305595398, "reward_std": 0.19222553074359894, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13873133063316345, "sampling/sampling_logp_difference/max": 1.80045747756958, "sampling/importance_sampling_ratio/min": 0.16522328555583954, "sampling/importance_sampling_ratio/mean": 1.0392733812332153, "sampling/importance_sampling_ratio/max": 1.945328712463379, "entropy": 0.7405487895011902, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.08850523363798857, "clip_ratio/high_max": 0.08850523363798857, "clip_ratio/region_mean": 0.08850523363798857, "reward_total_mean": 0.4702560305595398, "reward_meter_mean": 0.6961934566497803, "reward_meter_std": 0.27985823154449463, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.817654013633728, "reward_repeat_soft_std": 0.04878468066453934, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.13274572789669037, "reward_total_composite_mean": 0.4702560305595398, "reward_total_composite_std": 0.19222553074359894} {"timestamp_utc": "2026-04-13T10:47:28Z", "mode": "train", "global_step": 1390, "epoch": 0.13962832747363135, "loss": -0.0411, "grad_norm": 11.59184455871582, "learning_rate": 5.790909090909091e-06, "num_tokens": 2453881.0, "completions/mean_length": 38.25, "completions/min_length": 31.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.25, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.0587540939450264, "rewards/meter/std": 0.036901943385601044, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8439459800720215, "rewards/repeat_soft/std": 0.0702349916100502, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.3458341658115387, "rewards/total_composite/std": 0.016634918749332428, "reward": 0.3458341658115387, "reward_std": 0.01663491502404213, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09202395379543304, "sampling/sampling_logp_difference/max": 1.4838995933532715, "sampling/importance_sampling_ratio/min": 0.22675172984600067, "sampling/importance_sampling_ratio/mean": 1.01513671875, "sampling/importance_sampling_ratio/max": 1.8051742315292358, "entropy": 0.5906520001590252, "clip_ratio/low_mean": 0.027295285370200872, "clip_ratio/low_min": 0.027295285370200872, "clip_ratio/high_mean": 0.06077552307397127, "clip_ratio/high_max": 0.06077552307397127, "clip_ratio/region_mean": 0.08807080844417214, "reward_total_mean": 0.3458341658115387, "reward_meter_mean": 0.0587540939450264, "reward_meter_std": 0.036901943385601044, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8439459800720215, "reward_repeat_soft_std": 0.0702349916100502, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.3458341658115387, "reward_total_composite_std": 0.016634918749332428} {"timestamp_utc": "2026-04-13T10:47:34Z", "mode": "train", "global_step": 1391, "epoch": 0.13972877950778503, "loss": 0.0251, "grad_norm": 10.879137992858887, "learning_rate": 5.787878787878788e-06, "num_tokens": 2455597.0, "completions/mean_length": 43.5, "completions/min_length": 37.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.5, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9572920203208923, "rewards/meter/std": 0.021061226725578308, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7102234363555908, "rewards/repeat_soft/std": 0.09355749189853668, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.1249857097864151, "rewards/total_composite/mean": 0.5172368288040161, "rewards/total_composite/std": 0.07413040846586227, "reward": 0.5172368288040161, "reward_std": 0.07413038611412048, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08461681008338928, "sampling/sampling_logp_difference/max": 1.3330113887786865, "sampling/importance_sampling_ratio/min": 0.2636820375919342, "sampling/importance_sampling_ratio/mean": 1.00737464427948, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4708849526941776, "clip_ratio/low_mean": 0.025728388223797083, "clip_ratio/low_min": 0.025728388223797083, "clip_ratio/high_mean": 0.059031757060438395, "clip_ratio/high_max": 0.059031757060438395, "clip_ratio/region_mean": 0.08476014528423548, "reward_total_mean": 0.5172368288040161, "reward_meter_mean": 0.9572920203208923, "reward_meter_std": 0.021061226725578308, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7102234363555908, "reward_repeat_soft_std": 0.09355749189853668, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.1249857097864151, "reward_total_composite_mean": 0.5172368288040161, "reward_total_composite_std": 0.07413040846586227} {"timestamp_utc": "2026-04-13T10:47:40Z", "mode": "train", "global_step": 1392, "epoch": 0.13982923154193871, "loss": -0.0233, "grad_norm": 17.71356773376465, "learning_rate": 5.784848484848486e-06, "num_tokens": 2456934.0, "completions/mean_length": 21.125, "completions/min_length": 17.0, "completions/max_length": 27.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.125, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.8005595207214355, "rewards/meter/std": 0.3589307367801666, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9440689086914062, "rewards/repeat_soft/std": 0.029856212437152863, "rewards/judge_quality/mean": 0.6187499761581421, "rewards/judge_quality/std": 0.24976776540279388, "rewards/total_composite/mean": 0.6497457027435303, "rewards/total_composite/std": 0.20137529075145721, "reward": 0.6497457027435303, "reward_std": 0.20137529075145721, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11411852389574051, "sampling/sampling_logp_difference/max": 0.8509979248046875, "sampling/importance_sampling_ratio/min": 0.4269886016845703, "sampling/importance_sampling_ratio/mean": 1.0057789087295532, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8141906559467316, "clip_ratio/low_mean": 0.07073577027767897, "clip_ratio/low_min": 0.07073577027767897, "clip_ratio/high_mean": 0.017316017765551805, "clip_ratio/high_max": 0.017316017765551805, "clip_ratio/region_mean": 0.08805178804323077, "reward_total_mean": 0.6497457027435303, "reward_meter_mean": 0.8005595207214355, "reward_meter_std": 0.3589307367801666, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9440689086914062, "reward_repeat_soft_std": 0.029856212437152863, "reward_judge_quality_mean": 0.6187499761581421, "reward_judge_quality_std": 0.24976776540279388, "reward_total_composite_mean": 0.6497457027435303, "reward_total_composite_std": 0.20137529075145721} {"timestamp_utc": "2026-04-13T10:47:46Z", "mode": "train", "global_step": 1393, "epoch": 0.13992968357609242, "loss": 0.0317, "grad_norm": 22.41649627685547, "learning_rate": 5.781818181818181e-06, "num_tokens": 2458266.0, "completions/mean_length": 22.5, "completions/min_length": 17.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.5, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.8784810304641724, "rewards/meter/std": 0.2654203772544861, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8913486003875732, "rewards/repeat_soft/std": 0.07164983451366425, "rewards/judge_quality/mean": 0.4099999964237213, "rewards/judge_quality/std": 0.06633248925209045, "rewards/total_composite/mean": 0.565843939781189, "rewards/total_composite/std": 0.07566385716199875, "reward": 0.565843939781189, "reward_std": 0.07566386461257935, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13356241583824158, "sampling/sampling_logp_difference/max": 1.678497076034546, "sampling/importance_sampling_ratio/min": 0.18665429949760437, "sampling/importance_sampling_ratio/mean": 1.012643814086914, "sampling/importance_sampling_ratio/max": 1.7101576328277588, "entropy": 0.9071509465575218, "clip_ratio/low_mean": 0.03941993601620197, "clip_ratio/low_min": 0.03941993601620197, "clip_ratio/high_mean": 0.05967909004539251, "clip_ratio/high_max": 0.05967909004539251, "clip_ratio/region_mean": 0.09909902606159449, "reward_total_mean": 0.565843939781189, "reward_meter_mean": 0.8784810304641724, "reward_meter_std": 0.2654203772544861, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8913486003875732, "reward_repeat_soft_std": 0.07164983451366425, "reward_judge_quality_mean": 0.4099999964237213, "reward_judge_quality_std": 0.06633248925209045, "reward_total_composite_mean": 0.565843939781189, "reward_total_composite_std": 0.07566385716199875} {"timestamp_utc": "2026-04-13T10:47:52Z", "mode": "train", "global_step": 1394, "epoch": 0.1400301356102461, "loss": -0.0054, "grad_norm": 11.413646697998047, "learning_rate": 5.7787878787878795e-06, "num_tokens": 2459885.0, "completions/mean_length": 42.375, "completions/min_length": 36.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.375, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.9350890517234802, "rewards/meter/std": 0.12849915027618408, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8567144870758057, "rewards/repeat_soft/std": 0.08090715855360031, "rewards/judge_quality/mean": 0.4724999964237213, "rewards/judge_quality/std": 0.1011011004447937, "rewards/total_composite/mean": 0.6157437562942505, "rewards/total_composite/std": 0.07533933222293854, "reward": 0.6157437562942505, "reward_std": 0.07533932477235794, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13664910197257996, "sampling/sampling_logp_difference/max": 1.7999773025512695, "sampling/importance_sampling_ratio/min": 0.16530264914035797, "sampling/importance_sampling_ratio/mean": 0.99691241979599, "sampling/importance_sampling_ratio/max": 1.8642349243164062, "entropy": 0.7245934903621674, "clip_ratio/low_mean": 0.07677233032882214, "clip_ratio/low_min": 0.07677233032882214, "clip_ratio/high_mean": 0.020740161649882793, "clip_ratio/high_max": 0.020740161649882793, "clip_ratio/region_mean": 0.09751249197870493, "reward_total_mean": 0.6157437562942505, "reward_meter_mean": 0.9350890517234802, "reward_meter_std": 0.12849915027618408, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8567144870758057, "reward_repeat_soft_std": 0.08090715855360031, "reward_judge_quality_mean": 0.4724999964237213, "reward_judge_quality_std": 0.1011011004447937, "reward_total_composite_mean": 0.6157437562942505, "reward_total_composite_std": 0.07533933222293854} {"timestamp_utc": "2026-04-13T10:47:58Z", "mode": "train", "global_step": 1395, "epoch": 0.1401305876443998, "loss": -0.0168, "grad_norm": 9.653735160827637, "learning_rate": 5.775757575757577e-06, "num_tokens": 2461346.0, "completions/mean_length": 37.625, "completions/min_length": 34.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.625, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.9894571900367737, "rewards/meter/std": 0.009738119319081306, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7672044038772583, "rewards/repeat_soft/std": 0.07762787491083145, "rewards/judge_quality/mean": 0.24250000715255737, "rewards/judge_quality/std": 0.11792854964733124, "rewards/total_composite/mean": 0.47067469358444214, "rewards/total_composite/std": 0.08534766733646393, "reward": 0.47067469358444214, "reward_std": 0.08534765243530273, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09565913677215576, "sampling/sampling_logp_difference/max": 1.7760858535766602, "sampling/importance_sampling_ratio/min": 0.16929951310157776, "sampling/importance_sampling_ratio/mean": 1.0008505582809448, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5285905115306377, "clip_ratio/low_mean": 0.026665922487154603, "clip_ratio/low_min": 0.026665922487154603, "clip_ratio/high_mean": 0.0553113566711545, "clip_ratio/high_max": 0.0553113566711545, "clip_ratio/region_mean": 0.0819772791583091, "reward_total_mean": 0.47067469358444214, "reward_meter_mean": 0.9894571900367737, "reward_meter_std": 0.009738119319081306, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7672044038772583, "reward_repeat_soft_std": 0.07762787491083145, "reward_judge_quality_mean": 0.24250000715255737, "reward_judge_quality_std": 0.11792854964733124, "reward_total_composite_mean": 0.47067469358444214, "reward_total_composite_std": 0.08534766733646393} {"timestamp_utc": "2026-04-13T10:48:05Z", "mode": "train", "global_step": 1396, "epoch": 0.1402310396785535, "loss": 0.0376, "grad_norm": 11.208836555480957, "learning_rate": 5.772727272727273e-06, "num_tokens": 2462702.0, "completions/mean_length": 26.5, "completions/min_length": 21.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.5, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.25748664140701294, "rewards/meter/std": 0.43729761242866516, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.8025000095367432, "rewards/judge_quality/std": 0.21756774187088013, "rewards/total_composite/mean": 0.4982941150665283, "rewards/total_composite/std": 0.2615428566932678, "reward": 0.4982941150665283, "reward_std": 0.2615428566932678, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1218995675444603, "sampling/sampling_logp_difference/max": 1.2938337326049805, "sampling/importance_sampling_ratio/min": 0.27421748638153076, "sampling/importance_sampling_ratio/mean": 1.0340930223464966, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8053631111979485, "clip_ratio/low_mean": 0.11612405255436897, "clip_ratio/low_min": 0.11612405255436897, "clip_ratio/high_mean": 0.02425925899296999, "clip_ratio/high_max": 0.02425925899296999, "clip_ratio/region_mean": 0.14038331154733896, "reward_total_mean": 0.4982941150665283, "reward_meter_mean": 0.25748664140701294, "reward_meter_std": 0.43729761242866516, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.8025000095367432, "reward_judge_quality_std": 0.21756774187088013, "reward_total_composite_mean": 0.4982941150665283, "reward_total_composite_std": 0.2615428566932678} {"timestamp_utc": "2026-04-13T10:48:12Z", "mode": "train", "global_step": 1397, "epoch": 0.14033149171270717, "loss": -0.0281, "grad_norm": 7.8243865966796875, "learning_rate": 5.76969696969697e-06, "num_tokens": 2464378.0, "completions/mean_length": 44.5, "completions/min_length": 35.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.5, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9040476083755493, "rewards/meter/std": 0.21422027051448822, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8503400087356567, "rewards/repeat_soft/std": 0.05986413732171059, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.5766725540161133, "rewards/total_composite/std": 0.06210015341639519, "reward": 0.5766725540161133, "reward_std": 0.06210014224052429, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1364356279373169, "sampling/sampling_logp_difference/max": 2.3001978397369385, "sampling/importance_sampling_ratio/min": 0.10023900866508484, "sampling/importance_sampling_ratio/mean": 0.9939990043640137, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6070137321949005, "clip_ratio/low_mean": 0.016447369009256363, "clip_ratio/low_min": 0.016447369009256363, "clip_ratio/high_mean": 0.10732440184801817, "clip_ratio/high_max": 0.10732440184801817, "clip_ratio/region_mean": 0.12377177085727453, "reward_total_mean": 0.5766725540161133, "reward_meter_mean": 0.9040476083755493, "reward_meter_std": 0.21422027051448822, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8503400087356567, "reward_repeat_soft_std": 0.05986413732171059, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.5766725540161133, "reward_total_composite_std": 0.06210015341639519} {"timestamp_utc": "2026-04-13T10:48:18Z", "mode": "train", "global_step": 1398, "epoch": 0.14043194374686088, "loss": 0.0651, "grad_norm": 13.051623344421387, "learning_rate": 5.766666666666667e-06, "num_tokens": 2466024.0, "completions/mean_length": 37.75, "completions/min_length": 33.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.75, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.6293722987174988, "rewards/meter/std": 0.3659643232822418, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9026373624801636, "rewards/repeat_soft/std": 0.023215971887111664, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.5114091634750366, "rewards/total_composite/std": 0.09964381158351898, "reward": 0.5114091634750366, "reward_std": 0.09964380413293839, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15148068964481354, "sampling/sampling_logp_difference/max": 1.5002524852752686, "sampling/importance_sampling_ratio/min": 0.22307385504245758, "sampling/importance_sampling_ratio/mean": 0.9983512163162231, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.840466633439064, "clip_ratio/low_mean": 0.0661195544525981, "clip_ratio/low_min": 0.0661195544525981, "clip_ratio/high_mean": 0.06789673771709204, "clip_ratio/high_max": 0.06789673771709204, "clip_ratio/region_mean": 0.13401629216969013, "reward_total_mean": 0.5114091634750366, "reward_meter_mean": 0.6293722987174988, "reward_meter_std": 0.3659643232822418, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9026373624801636, "reward_repeat_soft_std": 0.023215971887111664, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.5114091634750366, "reward_total_composite_std": 0.09964381158351898} {"timestamp_utc": "2026-04-13T10:48:24Z", "mode": "train", "global_step": 1399, "epoch": 0.14053239578101456, "loss": -0.0003, "grad_norm": 9.11116886138916, "learning_rate": 5.763636363636365e-06, "num_tokens": 2467635.0, "completions/mean_length": 43.375, "completions/min_length": 39.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.375, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.9829332828521729, "rewards/meter/std": 0.010218682698905468, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8421238660812378, "rewards/repeat_soft/std": 0.06742725521326065, "rewards/judge_quality/mean": 0.4137499928474426, "rewards/judge_quality/std": 0.06781013309955597, "rewards/total_composite/mean": 0.5905901193618774, "rewards/total_composite/std": 0.04515933245420456, "reward": 0.5905901193618774, "reward_std": 0.045159339904785156, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11853578686714172, "sampling/sampling_logp_difference/max": 1.983877182006836, "sampling/importance_sampling_ratio/min": 0.137534961104393, "sampling/importance_sampling_ratio/mean": 1.008230447769165, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6134412474930286, "clip_ratio/low_mean": 0.026244589127600193, "clip_ratio/low_min": 0.026244589127600193, "clip_ratio/high_mean": 0.0931791178882122, "clip_ratio/high_max": 0.0931791178882122, "clip_ratio/region_mean": 0.1194237070158124, "reward_total_mean": 0.5905901193618774, "reward_meter_mean": 0.9829332828521729, "reward_meter_std": 0.010218682698905468, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8421238660812378, "reward_repeat_soft_std": 0.06742725521326065, "reward_judge_quality_mean": 0.4137499928474426, "reward_judge_quality_std": 0.06781013309955597, "reward_total_composite_mean": 0.5905901193618774, "reward_total_composite_std": 0.04515933245420456} {"timestamp_utc": "2026-04-13T10:48:35Z", "mode": "train", "global_step": 1400, "epoch": 0.14063284781516824, "loss": 0.0561, "grad_norm": 11.920273780822754, "learning_rate": 5.760606060606061e-06, "num_tokens": 2469500.0, "completions/mean_length": 61.125, "completions/min_length": 53.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.125, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9751566648483276, "rewards/meter/std": 0.01855885051190853, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8131760954856873, "rewards/repeat_soft/std": 0.07321101427078247, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5881941914558411, "rewards/total_composite/std": 0.012843296863138676, "reward": 0.5881941914558411, "reward_std": 0.01284329779446125, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14106321334838867, "sampling/sampling_logp_difference/max": 2.0515499114990234, "sampling/importance_sampling_ratio/min": 0.12853553891181946, "sampling/importance_sampling_ratio/mean": 0.9969595670700073, "sampling/importance_sampling_ratio/max": 1.8442491292953491, "entropy": 0.721710816025734, "clip_ratio/low_mean": 0.06279363110661507, "clip_ratio/low_min": 0.06279363110661507, "clip_ratio/high_mean": 0.07298974320292473, "clip_ratio/high_max": 0.07298974320292473, "clip_ratio/region_mean": 0.1357833743095398, "reward_total_mean": 0.5881941914558411, "reward_meter_mean": 0.9751566648483276, "reward_meter_std": 0.01855885051190853, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8131760954856873, "reward_repeat_soft_std": 0.07321101427078247, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5881941914558411, "reward_total_composite_std": 0.012843296863138676} {"timestamp_utc": "2026-04-13T10:49:13Z", "mode": "eval", "global_step": 1400, "epoch": 0.14063284781516824, "eval_loss": NaN, "eval_runtime": 38.0793, "eval_samples_per_second": 2.101, "eval_steps_per_second": 0.263, "eval_num_tokens": 2469500.0, "eval_completions/mean_length": 66.6375, "eval_completions/min_length": 32.7, "eval_completions/max_length": 109.7, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 66.6375, "eval_completions/min_terminated_length": 32.7, "eval_completions/max_terminated_length": 109.7, "eval_rewards/meter/mean": 0.8095608830451966, "eval_rewards/meter/std": 0.2641849434003234, "eval_rewards/count_adherence/mean": 0.96583331823349, "eval_rewards/count_adherence/std": 0.06645093113183975, "eval_rewards/hard_gate/mean": 0.9875, "eval_rewards/hard_gate/std": 0.03535533845424652, "eval_rewards/repeat_soft/mean": 0.8524089634418488, "eval_rewards/repeat_soft/std": 0.10649546608328819, "eval_rewards/judge_quality/mean": 0.45062499940395356, "eval_rewards/judge_quality/std": 0.15118036419153214, "eval_rewards/total_composite/mean": 0.5535447299480438, "eval_rewards/total_composite/std": 0.1373872399330139, "eval_reward": 0.5535447299480438, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.056691231578588484, "eval_sampling/sampling_logp_difference/max": 0.9773582935333252, "eval_sampling/importance_sampling_ratio/min": 0.3831986427307129, "eval_sampling/importance_sampling_ratio/mean": 1.0140406012535095, "eval_sampling/importance_sampling_ratio/max": 1.4029753446578979, "eval_entropy": 0.6156320989131927, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5535447299480438, "eval_reward_meter_mean": 0.8095608830451966, "eval_reward_meter_std": 0.2641849434003234, "eval_reward_count_adherence_mean": 0.96583331823349, "eval_reward_count_adherence_std": 0.06645093113183975, "eval_reward_hard_gate_mean": 0.9875, "eval_reward_hard_gate_std": 0.03535533845424652, "eval_reward_repeat_soft_mean": 0.8524089634418488, "eval_reward_repeat_soft_std": 0.10649546608328819, "eval_reward_judge_quality_mean": 0.45062499940395356, "eval_reward_judge_quality_std": 0.15118036419153214, "eval_reward_total_composite_mean": 0.5535447299480438, "eval_reward_total_composite_std": 0.1373872399330139} {"timestamp_utc": "2026-04-13T10:49:22Z", "mode": "train", "global_step": 1401, "epoch": 0.14073329984932195, "loss": 0.0659, "grad_norm": 17.8997802734375, "learning_rate": 5.7575757575757586e-06, "num_tokens": 2470883.0, "completions/mean_length": 37.875, "completions/min_length": 33.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.875, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.5351673364639282, "rewards/meter/std": 0.3454230725765228, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8383328914642334, "rewards/repeat_soft/std": 0.1236555278301239, "rewards/judge_quality/mean": 0.5112500190734863, "rewards/judge_quality/std": 0.268936425447464, "rewards/total_composite/mean": 0.5070319771766663, "rewards/total_composite/std": 0.17160125076770782, "reward": 0.5070319771766663, "reward_std": 0.17160125076770782, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11590097099542618, "sampling/sampling_logp_difference/max": 1.930302619934082, "sampling/importance_sampling_ratio/min": 0.1451042741537094, "sampling/importance_sampling_ratio/mean": 0.9949010610580444, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6245596781373024, "clip_ratio/low_mean": 0.04386756452731788, "clip_ratio/low_min": 0.04386756452731788, "clip_ratio/high_mean": 0.05764055158942938, "clip_ratio/high_max": 0.05764055158942938, "clip_ratio/region_mean": 0.10150811611674726, "reward_total_mean": 0.5070319771766663, "reward_meter_mean": 0.5351673364639282, "reward_meter_std": 0.3454230725765228, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8383328914642334, "reward_repeat_soft_std": 0.1236555278301239, "reward_judge_quality_mean": 0.5112500190734863, "reward_judge_quality_std": 0.268936425447464, "reward_total_composite_mean": 0.5070319771766663, "reward_total_composite_std": 0.17160125076770782} {"timestamp_utc": "2026-04-13T10:49:29Z", "mode": "train", "global_step": 1402, "epoch": 0.14083375188347563, "loss": 0.0051, "grad_norm": 10.03658390045166, "learning_rate": 5.754545454545455e-06, "num_tokens": 2472649.0, "completions/mean_length": 63.75, "completions/min_length": 58.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.75, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9593009948730469, "rewards/meter/std": 0.08032622188329697, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9324356317520142, "rewards/repeat_soft/std": 0.042951393872499466, "rewards/judge_quality/mean": 0.5149999856948853, "rewards/judge_quality/std": 0.1810288280248642, "rewards/total_composite/mean": 0.6627146005630493, "rewards/total_composite/std": 0.12697963416576385, "reward": 0.6627146005630493, "reward_std": 0.12697963416576385, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11571583896875381, "sampling/sampling_logp_difference/max": 3.0238711833953857, "sampling/importance_sampling_ratio/min": 0.04861266538500786, "sampling/importance_sampling_ratio/mean": 1.010787010192871, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.593965407460928, "clip_ratio/low_mean": 0.08435330679640174, "clip_ratio/low_min": 0.08435330679640174, "clip_ratio/high_mean": 0.03256302606314421, "clip_ratio/high_max": 0.03256302606314421, "clip_ratio/region_mean": 0.11691633285954595, "reward_total_mean": 0.6627146005630493, "reward_meter_mean": 0.9593009948730469, "reward_meter_std": 0.08032622188329697, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9324356317520142, "reward_repeat_soft_std": 0.042951393872499466, "reward_judge_quality_mean": 0.5149999856948853, "reward_judge_quality_std": 0.1810288280248642, "reward_total_composite_mean": 0.6627146005630493, "reward_total_composite_std": 0.12697963416576385} {"timestamp_utc": "2026-04-13T10:49:36Z", "mode": "train", "global_step": 1403, "epoch": 0.14093420391762934, "loss": -0.011, "grad_norm": 15.948739051818848, "learning_rate": 5.751515151515152e-06, "num_tokens": 2474383.0, "completions/mean_length": 36.75, "completions/min_length": 32.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.75, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.9653463363647461, "rewards/meter/std": 0.028619877994060516, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8667511343955994, "rewards/repeat_soft/std": 0.09769817441701889, "rewards/judge_quality/mean": 0.4050000011920929, "rewards/judge_quality/std": 0.10392304509878159, "rewards/total_composite/mean": 0.5837519764900208, "rewards/total_composite/std": 0.07472193986177444, "reward": 0.5837519764900208, "reward_std": 0.07472193986177444, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12553322315216064, "sampling/sampling_logp_difference/max": 1.3544254302978516, "sampling/importance_sampling_ratio/min": 0.25809556245803833, "sampling/importance_sampling_ratio/mean": 1.013904094696045, "sampling/importance_sampling_ratio/max": 1.9975882768630981, "entropy": 0.6746665462851524, "clip_ratio/low_mean": 0.0234375, "clip_ratio/low_min": 0.0234375, "clip_ratio/high_mean": 0.09110360452905297, "clip_ratio/high_max": 0.09110360452905297, "clip_ratio/region_mean": 0.11454110452905297, "reward_total_mean": 0.5837519764900208, "reward_meter_mean": 0.9653463363647461, "reward_meter_std": 0.028619877994060516, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8667511343955994, "reward_repeat_soft_std": 0.09769817441701889, "reward_judge_quality_mean": 0.4050000011920929, "reward_judge_quality_std": 0.10392304509878159, "reward_total_composite_mean": 0.5837519764900208, "reward_total_composite_std": 0.07472193986177444} {"timestamp_utc": "2026-04-13T10:49:44Z", "mode": "train", "global_step": 1404, "epoch": 0.14103465595178302, "loss": 0.0233, "grad_norm": 10.28912353515625, "learning_rate": 5.748484848484849e-06, "num_tokens": 2476611.0, "completions/mean_length": 78.5, "completions/min_length": 70.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.5, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.5133029222488403, "rewards/meter/std": 0.36811619997024536, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8736454844474792, "rewards/repeat_soft/std": 0.06013016775250435, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.5057510137557983, "rewards/total_composite/std": 0.1692267805337906, "reward": 0.5057510137557983, "reward_std": 0.1692267656326294, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11171042174100876, "sampling/sampling_logp_difference/max": 2.0017662048339844, "sampling/importance_sampling_ratio/min": 0.1350964605808258, "sampling/importance_sampling_ratio/mean": 1.0089904069900513, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4880719818174839, "clip_ratio/low_mean": 0.05477775679901242, "clip_ratio/low_min": 0.05477775679901242, "clip_ratio/high_mean": 0.033478156197816133, "clip_ratio/high_max": 0.033478156197816133, "clip_ratio/region_mean": 0.08825591299682856, "reward_total_mean": 0.5057510137557983, "reward_meter_mean": 0.5133029222488403, "reward_meter_std": 0.36811619997024536, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8736454844474792, "reward_repeat_soft_std": 0.06013016775250435, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.5057510137557983, "reward_total_composite_std": 0.1692267805337906} {"timestamp_utc": "2026-04-13T10:49:50Z", "mode": "train", "global_step": 1405, "epoch": 0.1411351079859367, "loss": 0.0774, "grad_norm": 14.689164161682129, "learning_rate": 5.745454545454546e-06, "num_tokens": 2478191.0, "completions/mean_length": 40.5, "completions/min_length": 34.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.5, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.7688623070716858, "rewards/meter/std": 0.3516474962234497, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8841646909713745, "rewards/repeat_soft/std": 0.08102813363075256, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.5471103191375732, "rewards/total_composite/std": 0.09307053685188293, "reward": 0.5471103191375732, "reward_std": 0.09307053685188293, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1309376060962677, "sampling/sampling_logp_difference/max": 1.5793876647949219, "sampling/importance_sampling_ratio/min": 0.20610125362873077, "sampling/importance_sampling_ratio/mean": 1.02289879322052, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8622675091028214, "clip_ratio/low_mean": 0.03683094959706068, "clip_ratio/low_min": 0.03683094959706068, "clip_ratio/high_mean": 0.10103944409638643, "clip_ratio/high_max": 0.10103944409638643, "clip_ratio/region_mean": 0.1378703936934471, "reward_total_mean": 0.5471103191375732, "reward_meter_mean": 0.7688623070716858, "reward_meter_std": 0.3516474962234497, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8841646909713745, "reward_repeat_soft_std": 0.08102813363075256, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.5471103191375732, "reward_total_composite_std": 0.09307053685188293} {"timestamp_utc": "2026-04-13T10:49:57Z", "mode": "train", "global_step": 1406, "epoch": 0.1412355600200904, "loss": 0.0302, "grad_norm": 12.480927467346191, "learning_rate": 5.742424242424242e-06, "num_tokens": 2480060.0, "completions/mean_length": 58.625, "completions/min_length": 48.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.625, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.7705498933792114, "rewards/meter/std": 0.32462453842163086, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7663149237632751, "rewards/repeat_soft/std": 0.0751347616314888, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5253073573112488, "rewards/total_composite/std": 0.08485578000545502, "reward": 0.5253073573112488, "reward_std": 0.08485576510429382, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11231084167957306, "sampling/sampling_logp_difference/max": 1.8563451766967773, "sampling/importance_sampling_ratio/min": 0.15624262392520905, "sampling/importance_sampling_ratio/mean": 0.9956095218658447, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5471113361418247, "clip_ratio/low_mean": 0.04088320955634117, "clip_ratio/low_min": 0.04088320955634117, "clip_ratio/high_mean": 0.07989784982055426, "clip_ratio/high_max": 0.07989784982055426, "clip_ratio/region_mean": 0.12078105937689543, "reward_total_mean": 0.5253073573112488, "reward_meter_mean": 0.7705498933792114, "reward_meter_std": 0.32462453842163086, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7663149237632751, "reward_repeat_soft_std": 0.0751347616314888, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5253073573112488, "reward_total_composite_std": 0.08485578000545502} {"timestamp_utc": "2026-04-13T10:50:02Z", "mode": "train", "global_step": 1407, "epoch": 0.1413360120542441, "loss": 0.0341, "grad_norm": 14.56716537475586, "learning_rate": 5.73939393939394e-06, "num_tokens": 2481489.0, "completions/mean_length": 23.625, "completions/min_length": 22.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.625, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.5254626870155334, "rewards/meter/std": 0.3642630875110626, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.951632022857666, "rewards/repeat_soft/std": 0.025855133309960365, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.10966669768095016, "rewards/total_composite/mean": 0.44512939453125, "rewards/total_composite/std": 0.0740780457854271, "reward": 0.44512939453125, "reward_std": 0.0740780457854271, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1347288340330124, "sampling/sampling_logp_difference/max": 1.4268412590026855, "sampling/importance_sampling_ratio/min": 0.2400660365819931, "sampling/importance_sampling_ratio/mean": 1.0078125, "sampling/importance_sampling_ratio/max": 1.599117636680603, "entropy": 0.8280720114707947, "clip_ratio/low_mean": 0.055887049064040184, "clip_ratio/low_min": 0.055887049064040184, "clip_ratio/high_mean": 0.043560607358813286, "clip_ratio/high_max": 0.043560607358813286, "clip_ratio/region_mean": 0.09944765642285347, "reward_total_mean": 0.44512939453125, "reward_meter_mean": 0.5254626870155334, "reward_meter_std": 0.3642630875110626, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.951632022857666, "reward_repeat_soft_std": 0.025855133309960365, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.10966669768095016, "reward_total_composite_mean": 0.44512939453125, "reward_total_composite_std": 0.0740780457854271} {"timestamp_utc": "2026-04-13T10:50:09Z", "mode": "train", "global_step": 1408, "epoch": 0.1414364640883978, "loss": -0.0026, "grad_norm": 12.827202796936035, "learning_rate": 5.736363636363637e-06, "num_tokens": 2483058.0, "completions/mean_length": 40.125, "completions/min_length": 29.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.125, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.715053141117096, "rewards/meter/std": 0.29322072863578796, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9744253158569336, "rewards/repeat_soft/std": 0.020732590928673744, "rewards/judge_quality/mean": 0.565000057220459, "rewards/judge_quality/std": 0.194054514169693, "rewards/total_composite/mean": 0.6142382621765137, "rewards/total_composite/std": 0.14438693225383759, "reward": 0.6142382621765137, "reward_std": 0.14438693225383759, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15014910697937012, "sampling/sampling_logp_difference/max": 1.556006908416748, "sampling/importance_sampling_ratio/min": 0.21097683906555176, "sampling/importance_sampling_ratio/mean": 0.9943320751190186, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5537070706486702, "clip_ratio/low_mean": 0.06867803446948528, "clip_ratio/low_min": 0.06867803446948528, "clip_ratio/high_mean": 0.05681167170405388, "clip_ratio/high_max": 0.05681167170405388, "clip_ratio/region_mean": 0.12548970617353916, "reward_total_mean": 0.6142382621765137, "reward_meter_mean": 0.715053141117096, "reward_meter_std": 0.29322072863578796, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9744253158569336, "reward_repeat_soft_std": 0.020732590928673744, "reward_judge_quality_mean": 0.565000057220459, "reward_judge_quality_std": 0.194054514169693, "reward_total_composite_mean": 0.6142382621765137, "reward_total_composite_std": 0.14438693225383759} {"timestamp_utc": "2026-04-13T10:50:16Z", "mode": "train", "global_step": 1409, "epoch": 0.14153691612255148, "loss": -0.0022, "grad_norm": 12.561360359191895, "learning_rate": 5.733333333333334e-06, "num_tokens": 2485080.0, "completions/mean_length": 74.75, "completions/min_length": 71.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.75, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.8560695648193359, "rewards/meter/std": 0.27132144570350647, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9129428863525391, "rewards/repeat_soft/std": 0.034970927983522415, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.510246992111206, "rewards/total_composite/std": 0.21708330512046814, "reward": 0.510246992111206, "reward_std": 0.21708329021930695, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12822890281677246, "sampling/sampling_logp_difference/max": 2.5206103324890137, "sampling/importance_sampling_ratio/min": 0.20428669452667236, "sampling/importance_sampling_ratio/mean": 1.0028680562973022, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6156154870986938, "clip_ratio/low_mean": 0.010416666977107525, "clip_ratio/low_min": 0.010416666977107525, "clip_ratio/high_mean": 0.09973405580967665, "clip_ratio/high_max": 0.09973405580967665, "clip_ratio/region_mean": 0.11015072278678417, "reward_total_mean": 0.510246992111206, "reward_meter_mean": 0.8560695648193359, "reward_meter_std": 0.27132144570350647, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9129428863525391, "reward_repeat_soft_std": 0.034970927983522415, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.510246992111206, "reward_total_composite_std": 0.21708330512046814} {"timestamp_utc": "2026-04-13T10:50:22Z", "mode": "train", "global_step": 1410, "epoch": 0.14163736815670516, "loss": -0.053, "grad_norm": 14.60198974609375, "learning_rate": 5.7303030303030305e-06, "num_tokens": 2486748.0, "completions/mean_length": 40.5, "completions/min_length": 30.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.5, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.8249857425689697, "rewards/meter/std": 0.33494699001312256, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.950890064239502, "rewards/repeat_soft/std": 0.051062971353530884, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.21357084810733795, "rewards/total_composite/mean": 0.6086096167564392, "rewards/total_composite/std": 0.15268997848033905, "reward": 0.6086096167564392, "reward_std": 0.15268996357917786, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16327844560146332, "sampling/sampling_logp_difference/max": 3.4548776149749756, "sampling/importance_sampling_ratio/min": 0.03159116953611374, "sampling/importance_sampling_ratio/mean": 0.9968622326850891, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8403591066598892, "clip_ratio/low_mean": 0.048809911124408245, "clip_ratio/low_min": 0.048809911124408245, "clip_ratio/high_mean": 0.06952887680381536, "clip_ratio/high_max": 0.06952887680381536, "clip_ratio/region_mean": 0.11833878792822361, "reward_total_mean": 0.6086096167564392, "reward_meter_mean": 0.8249857425689697, "reward_meter_std": 0.33494699001312256, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.950890064239502, "reward_repeat_soft_std": 0.051062971353530884, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.21357084810733795, "reward_total_composite_mean": 0.6086096167564392, "reward_total_composite_std": 0.15268997848033905} {"timestamp_utc": "2026-04-13T10:50:29Z", "mode": "train", "global_step": 1411, "epoch": 0.14173782019085887, "loss": 0.0223, "grad_norm": 20.06722640991211, "learning_rate": 5.727272727272728e-06, "num_tokens": 2488469.0, "completions/mean_length": 45.125, "completions/min_length": 42.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.125, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.75612473487854, "rewards/meter/std": 0.3957335948944092, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9366524815559387, "rewards/repeat_soft/std": 0.07409165799617767, "rewards/judge_quality/mean": 0.5362499952316284, "rewards/judge_quality/std": 0.16070716083049774, "rewards/total_composite/mean": 0.5985976457595825, "rewards/total_composite/std": 0.16009141504764557, "reward": 0.5985976457595825, "reward_std": 0.16009140014648438, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11883072555065155, "sampling/sampling_logp_difference/max": 2.6178369522094727, "sampling/importance_sampling_ratio/min": 0.0729605108499527, "sampling/importance_sampling_ratio/mean": 1.0040268898010254, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7764797955751419, "clip_ratio/low_mean": 0.027832756750285625, "clip_ratio/low_min": 0.027832756750285625, "clip_ratio/high_mean": 0.09952107258141041, "clip_ratio/high_max": 0.09952107258141041, "clip_ratio/region_mean": 0.12735382933169603, "reward_total_mean": 0.5985976457595825, "reward_meter_mean": 0.75612473487854, "reward_meter_std": 0.3957335948944092, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9366524815559387, "reward_repeat_soft_std": 0.07409165799617767, "reward_judge_quality_mean": 0.5362499952316284, "reward_judge_quality_std": 0.16070716083049774, "reward_total_composite_mean": 0.5985976457595825, "reward_total_composite_std": 0.16009141504764557} {"timestamp_utc": "2026-04-13T10:50:36Z", "mode": "train", "global_step": 1412, "epoch": 0.14183827222501255, "loss": 0.0228, "grad_norm": 19.718990325927734, "learning_rate": 5.724242424242424e-06, "num_tokens": 2490611.0, "completions/mean_length": 94.75, "completions/min_length": 80.0, "completions/max_length": 110.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.75, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 110.0, "rewards/meter/mean": 0.6265571713447571, "rewards/meter/std": 0.35264143347740173, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9424885511398315, "rewards/repeat_soft/std": 0.036095939576625824, "rewards/judge_quality/mean": 0.5324999690055847, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.5623034834861755, "rewards/total_composite/std": 0.15559940040111542, "reward": 0.5623034834861755, "reward_std": 0.15559938549995422, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13852238655090332, "sampling/sampling_logp_difference/max": 3.2659621238708496, "sampling/importance_sampling_ratio/min": 0.03816020488739014, "sampling/importance_sampling_ratio/mean": 1.0092533826828003, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7166869342327118, "clip_ratio/low_mean": 0.06569533981382847, "clip_ratio/low_min": 0.06569533981382847, "clip_ratio/high_mean": 0.057803040370345116, "clip_ratio/high_max": 0.057803040370345116, "clip_ratio/region_mean": 0.12349838018417358, "reward_total_mean": 0.5623034834861755, "reward_meter_mean": 0.6265571713447571, "reward_meter_std": 0.35264143347740173, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9424885511398315, "reward_repeat_soft_std": 0.036095939576625824, "reward_judge_quality_mean": 0.5324999690055847, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.5623034834861755, "reward_total_composite_std": 0.15559940040111542} {"timestamp_utc": "2026-04-13T10:50:44Z", "mode": "train", "global_step": 1413, "epoch": 0.14193872425916626, "loss": 0.0103, "grad_norm": 9.716988563537598, "learning_rate": 5.721212121212122e-06, "num_tokens": 2492560.0, "completions/mean_length": 68.625, "completions/min_length": 61.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.625, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9647419452667236, "rewards/meter/std": 0.047957222908735275, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7776961326599121, "rewards/repeat_soft/std": 0.05703750625252724, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.5277512669563293, "rewards/total_composite/std": 0.06632345169782639, "reward": 0.5277512669563293, "reward_std": 0.06632344424724579, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11125600337982178, "sampling/sampling_logp_difference/max": 1.4070215225219727, "sampling/importance_sampling_ratio/min": 0.24487155675888062, "sampling/importance_sampling_ratio/mean": 1.0192159414291382, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7872535884380341, "clip_ratio/low_mean": 0.043009101413190365, "clip_ratio/low_min": 0.043009101413190365, "clip_ratio/high_mean": 0.05227456334978342, "clip_ratio/high_max": 0.05227456334978342, "clip_ratio/region_mean": 0.09528366476297379, "reward_total_mean": 0.5277512669563293, "reward_meter_mean": 0.9647419452667236, "reward_meter_std": 0.047957222908735275, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7776961326599121, "reward_repeat_soft_std": 0.05703750625252724, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.5277512669563293, "reward_total_composite_std": 0.06632345169782639} {"timestamp_utc": "2026-04-13T10:50:50Z", "mode": "train", "global_step": 1414, "epoch": 0.14203917629331994, "loss": -0.0177, "grad_norm": 14.703536033630371, "learning_rate": 5.718181818181819e-06, "num_tokens": 2494106.0, "completions/mean_length": 46.25, "completions/min_length": 36.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.25, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.301479697227478, "rewards/meter/std": 0.358011394739151, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9754935503005981, "rewards/repeat_soft/std": 0.029871217906475067, "rewards/judge_quality/mean": 0.6112500429153442, "rewards/judge_quality/std": 0.25587037205696106, "rewards/total_composite/mean": 0.47409993410110474, "rewards/total_composite/std": 0.1980825662612915, "reward": 0.47409993410110474, "reward_std": 0.1980825513601303, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16618722677230835, "sampling/sampling_logp_difference/max": 1.4385216236114502, "sampling/importance_sampling_ratio/min": 0.2372782826423645, "sampling/importance_sampling_ratio/mean": 1.0048731565475464, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8378394320607185, "clip_ratio/low_mean": 0.0703724012710154, "clip_ratio/low_min": 0.0703724012710154, "clip_ratio/high_mean": 0.06799311935901642, "clip_ratio/high_max": 0.06799311935901642, "clip_ratio/region_mean": 0.13836552063003182, "reward_total_mean": 0.47409993410110474, "reward_meter_mean": 0.301479697227478, "reward_meter_std": 0.358011394739151, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9754935503005981, "reward_repeat_soft_std": 0.029871217906475067, "reward_judge_quality_mean": 0.6112500429153442, "reward_judge_quality_std": 0.25587037205696106, "reward_total_composite_mean": 0.47409993410110474, "reward_total_composite_std": 0.1980825662612915} {"timestamp_utc": "2026-04-13T10:50:57Z", "mode": "train", "global_step": 1415, "epoch": 0.14213962832747362, "loss": 0.013, "grad_norm": 13.053173065185547, "learning_rate": 5.715151515151516e-06, "num_tokens": 2495743.0, "completions/mean_length": 47.625, "completions/min_length": 43.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.625, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.8571093082427979, "rewards/meter/std": 0.27939456701278687, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9783151149749756, "rewards/repeat_soft/std": 0.0327603854238987, "rewards/judge_quality/mean": 0.5687500238418579, "rewards/judge_quality/std": 0.19773270189762115, "rewards/total_composite/mean": 0.6719194650650024, "rewards/total_composite/std": 0.1636134833097458, "reward": 0.6719194650650024, "reward_std": 0.1636134833097458, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11143966019153595, "sampling/sampling_logp_difference/max": 1.3081310987472534, "sampling/importance_sampling_ratio/min": 0.27032479643821716, "sampling/importance_sampling_ratio/mean": 1.005942702293396, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7236667647957802, "clip_ratio/low_mean": 0.08082544151693583, "clip_ratio/low_min": 0.08082544151693583, "clip_ratio/high_mean": 0.05500944051891565, "clip_ratio/high_max": 0.05500944051891565, "clip_ratio/region_mean": 0.13583488203585148, "reward_total_mean": 0.6719194650650024, "reward_meter_mean": 0.8571093082427979, "reward_meter_std": 0.27939456701278687, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9783151149749756, "reward_repeat_soft_std": 0.0327603854238987, "reward_judge_quality_mean": 0.5687500238418579, "reward_judge_quality_std": 0.19773270189762115, "reward_total_composite_mean": 0.6719194650650024, "reward_total_composite_std": 0.1636134833097458} {"timestamp_utc": "2026-04-13T10:51:03Z", "mode": "train", "global_step": 1416, "epoch": 0.14224008036162733, "loss": 0.0465, "grad_norm": 27.83479881286621, "learning_rate": 5.712121212121212e-06, "num_tokens": 2497121.0, "completions/mean_length": 20.25, "completions/min_length": 18.0, "completions/max_length": 23.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 20.25, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 23.0, "rewards/meter/mean": 0.9716972708702087, "rewards/meter/std": 0.035730406641960144, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.928876519203186, "rewards/repeat_soft/std": 0.08303026854991913, "rewards/judge_quality/mean": 0.4024999737739563, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.593076229095459, "rewards/total_composite/std": 0.03745569661259651, "reward": 0.593076229095459, "reward_std": 0.03745569661259651, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13063624501228333, "sampling/sampling_logp_difference/max": 1.1894512176513672, "sampling/importance_sampling_ratio/min": 0.30438825488090515, "sampling/importance_sampling_ratio/mean": 1.0124963521957397, "sampling/importance_sampling_ratio/max": 1.8049801588058472, "entropy": 0.6687679588794708, "clip_ratio/low_mean": 0.022997836116701365, "clip_ratio/low_min": 0.022997836116701365, "clip_ratio/high_mean": 0.057428672444075346, "clip_ratio/high_max": 0.057428672444075346, "clip_ratio/region_mean": 0.08042650856077671, "reward_total_mean": 0.593076229095459, "reward_meter_mean": 0.9716972708702087, "reward_meter_std": 0.035730406641960144, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.928876519203186, "reward_repeat_soft_std": 0.08303026854991913, "reward_judge_quality_mean": 0.4024999737739563, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.593076229095459, "reward_total_composite_std": 0.03745569661259651} {"timestamp_utc": "2026-04-13T10:51:09Z", "mode": "train", "global_step": 1417, "epoch": 0.142340532395781, "loss": 0.0011, "grad_norm": 14.798911094665527, "learning_rate": 5.7090909090909096e-06, "num_tokens": 2498561.0, "completions/mean_length": 30.0, "completions/min_length": 28.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.0, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9242525100708008, "rewards/meter/std": 0.07899387180805206, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8455250859260559, "rewards/repeat_soft/std": 0.04145512357354164, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.6020537614822388, "rewards/total_composite/std": 0.11468619108200073, "reward": 0.6020537614822388, "reward_std": 0.11468618363142014, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0789555236697197, "sampling/sampling_logp_difference/max": 1.3896409273147583, "sampling/importance_sampling_ratio/min": 0.24916476011276245, "sampling/importance_sampling_ratio/mean": 1.0224395990371704, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5047766417264938, "clip_ratio/low_mean": 0.046301571652293205, "clip_ratio/low_min": 0.046301571652293205, "clip_ratio/high_mean": 0.01953125, "clip_ratio/high_max": 0.01953125, "clip_ratio/region_mean": 0.0658328216522932, "reward_total_mean": 0.6020537614822388, "reward_meter_mean": 0.9242525100708008, "reward_meter_std": 0.07899387180805206, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8455250859260559, "reward_repeat_soft_std": 0.04145512357354164, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.6020537614822388, "reward_total_composite_std": 0.11468619108200073} {"timestamp_utc": "2026-04-13T10:51:15Z", "mode": "train", "global_step": 1418, "epoch": 0.14244098442993472, "loss": 0.0091, "grad_norm": 15.589071273803711, "learning_rate": 5.706060606060606e-06, "num_tokens": 2500117.0, "completions/mean_length": 43.5, "completions/min_length": 33.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.5, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.5782667398452759, "rewards/meter/std": 0.4150027632713318, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9840254783630371, "rewards/repeat_soft/std": 0.011035393923521042, "rewards/judge_quality/mean": 0.5475000143051147, "rewards/judge_quality/std": 0.14320316910743713, "rewards/total_composite/mean": 0.5291829109191895, "rewards/total_composite/std": 0.10961486399173737, "reward": 0.5291829109191895, "reward_std": 0.10961487144231796, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11890725791454315, "sampling/sampling_logp_difference/max": 1.7466232776641846, "sampling/importance_sampling_ratio/min": 0.1743617206811905, "sampling/importance_sampling_ratio/mean": 1.0020751953125, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7522838935256004, "clip_ratio/low_mean": 0.04423077031970024, "clip_ratio/low_min": 0.04423077031970024, "clip_ratio/high_mean": 0.06876990105956793, "clip_ratio/high_max": 0.06876990105956793, "clip_ratio/region_mean": 0.11300067137926817, "reward_total_mean": 0.5291829109191895, "reward_meter_mean": 0.5782667398452759, "reward_meter_std": 0.4150027632713318, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9840254783630371, "reward_repeat_soft_std": 0.011035393923521042, "reward_judge_quality_mean": 0.5475000143051147, "reward_judge_quality_std": 0.14320316910743713, "reward_total_composite_mean": 0.5291829109191895, "reward_total_composite_std": 0.10961486399173737} {"timestamp_utc": "2026-04-13T10:51:21Z", "mode": "train", "global_step": 1419, "epoch": 0.1425414364640884, "loss": 0.0123, "grad_norm": 18.427335739135742, "learning_rate": 5.703030303030303e-06, "num_tokens": 2501696.0, "completions/mean_length": 23.375, "completions/min_length": 20.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.375, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.9696396589279175, "rewards/meter/std": 0.03659556433558464, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.948153018951416, "rewards/repeat_soft/std": 0.0210404172539711, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.22720351815223694, "rewards/total_composite/mean": 0.6885734796524048, "rewards/total_composite/std": 0.13805341720581055, "reward": 0.6885734796524048, "reward_std": 0.13805341720581055, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11416921764612198, "sampling/sampling_logp_difference/max": 1.1906063556671143, "sampling/importance_sampling_ratio/min": 0.31950920820236206, "sampling/importance_sampling_ratio/mean": 1.0270253419876099, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5909772589802742, "clip_ratio/low_mean": 0.07892339676618576, "clip_ratio/low_min": 0.07892339676618576, "clip_ratio/high_mean": 0.02116402145475149, "clip_ratio/high_max": 0.02116402145475149, "clip_ratio/region_mean": 0.10008741822093725, "reward_total_mean": 0.6885734796524048, "reward_meter_mean": 0.9696396589279175, "reward_meter_std": 0.03659556433558464, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.948153018951416, "reward_repeat_soft_std": 0.0210404172539711, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.22720351815223694, "reward_total_composite_mean": 0.6885734796524048, "reward_total_composite_std": 0.13805341720581055} {"timestamp_utc": "2026-04-13T10:51:27Z", "mode": "train", "global_step": 1420, "epoch": 0.14264188849824208, "loss": 0.0516, "grad_norm": 12.975893020629883, "learning_rate": 5.7e-06, "num_tokens": 2503264.0, "completions/mean_length": 42.0, "completions/min_length": 37.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.0, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.9350648522377014, "rewards/meter/std": 0.08645668625831604, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9639267921447754, "rewards/repeat_soft/std": 0.03511298447847366, "rewards/judge_quality/mean": 0.6150000095367432, "rewards/judge_quality/std": 0.25286927819252014, "rewards/total_composite/mean": 0.7239274978637695, "rewards/total_composite/std": 0.1742723286151886, "reward": 0.7239274978637695, "reward_std": 0.1742723137140274, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12830279767513275, "sampling/sampling_logp_difference/max": 1.9506807327270508, "sampling/importance_sampling_ratio/min": 0.1421772539615631, "sampling/importance_sampling_ratio/mean": 1.0142592191696167, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8880967274308205, "clip_ratio/low_mean": 0.08618996618315578, "clip_ratio/low_min": 0.08618996618315578, "clip_ratio/high_mean": 0.043819112004712224, "clip_ratio/high_max": 0.043819112004712224, "clip_ratio/region_mean": 0.130009078187868, "reward_total_mean": 0.7239274978637695, "reward_meter_mean": 0.9350648522377014, "reward_meter_std": 0.08645668625831604, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9639267921447754, "reward_repeat_soft_std": 0.03511298447847366, "reward_judge_quality_mean": 0.6150000095367432, "reward_judge_quality_std": 0.25286927819252014, "reward_total_composite_mean": 0.7239274978637695, "reward_total_composite_std": 0.1742723286151886} {"timestamp_utc": "2026-04-13T10:51:33Z", "mode": "train", "global_step": 1421, "epoch": 0.1427423405323958, "loss": 0.001, "grad_norm": 14.406489372253418, "learning_rate": 5.696969696969698e-06, "num_tokens": 2504573.0, "completions/mean_length": 21.625, "completions/min_length": 19.0, "completions/max_length": 25.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.625, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 25.0, "rewards/meter/mean": 0.8017401695251465, "rewards/meter/std": 0.21783582866191864, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9435555934906006, "rewards/repeat_soft/std": 0.036296285688877106, "rewards/judge_quality/mean": 0.44749999046325684, "rewards/judge_quality/std": 0.20824094116687775, "rewards/total_composite/mean": 0.4621683955192566, "rewards/total_composite/std": 0.2980007827281952, "reward": 0.4621683955192566, "reward_std": 0.2980007827281952, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12893855571746826, "sampling/sampling_logp_difference/max": 0.9362697601318359, "sampling/importance_sampling_ratio/min": 0.3920876979827881, "sampling/importance_sampling_ratio/mean": 0.997369110584259, "sampling/importance_sampling_ratio/max": 1.6800825595855713, "entropy": 0.858201414346695, "clip_ratio/low_mean": 0.023863636888563633, "clip_ratio/low_min": 0.023863636888563633, "clip_ratio/high_mean": 0.13010139018297195, "clip_ratio/high_max": 0.13010139018297195, "clip_ratio/region_mean": 0.1539650270715356, "reward_total_mean": 0.4621683955192566, "reward_meter_mean": 0.8017401695251465, "reward_meter_std": 0.21783582866191864, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9435555934906006, "reward_repeat_soft_std": 0.036296285688877106, "reward_judge_quality_mean": 0.44749999046325684, "reward_judge_quality_std": 0.20824094116687775, "reward_total_composite_mean": 0.4621683955192566, "reward_total_composite_std": 0.2980007827281952} {"timestamp_utc": "2026-04-13T10:51:40Z", "mode": "train", "global_step": 1422, "epoch": 0.14284279256654947, "loss": 0.0346, "grad_norm": 9.34439754486084, "learning_rate": 5.693939393939394e-06, "num_tokens": 2506629.0, "completions/mean_length": 88.0, "completions/min_length": 83.0, "completions/max_length": 95.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 88.0, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.7999811172485352, "rewards/meter/std": 0.18497784435749054, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8585450053215027, "rewards/repeat_soft/std": 0.036236103624105453, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.5709871053695679, "rewards/total_composite/std": 0.09682025015354156, "reward": 0.5709871053695679, "reward_std": 0.09682025015354156, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1265259087085724, "sampling/sampling_logp_difference/max": 1.78059720993042, "sampling/importance_sampling_ratio/min": 0.1685374677181244, "sampling/importance_sampling_ratio/mean": 1.015747308731079, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6572273150086403, "clip_ratio/low_mean": 0.07174063427373767, "clip_ratio/low_min": 0.07174063427373767, "clip_ratio/high_mean": 0.04543060716241598, "clip_ratio/high_max": 0.04543060716241598, "clip_ratio/region_mean": 0.11717124143615365, "reward_total_mean": 0.5709871053695679, "reward_meter_mean": 0.7999811172485352, "reward_meter_std": 0.18497784435749054, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8585450053215027, "reward_repeat_soft_std": 0.036236103624105453, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.5709871053695679, "reward_total_composite_std": 0.09682025015354156} {"timestamp_utc": "2026-04-13T10:51:47Z", "mode": "train", "global_step": 1423, "epoch": 0.14294324460070315, "loss": 0.0202, "grad_norm": 8.207839965820312, "learning_rate": 5.690909090909091e-06, "num_tokens": 2508931.0, "completions/mean_length": 96.75, "completions/min_length": 77.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.75, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.7195549607276917, "rewards/meter/std": 0.3614080846309662, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9357661008834839, "rewards/repeat_soft/std": 0.04376447945833206, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.5204696655273438, "rewards/total_composite/std": 0.13906559348106384, "reward": 0.5204696655273438, "reward_std": 0.13906559348106384, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1431320309638977, "sampling/sampling_logp_difference/max": 4.391334533691406, "sampling/importance_sampling_ratio/min": 0.012384191155433655, "sampling/importance_sampling_ratio/mean": 1.0087347030639648, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7626270726323128, "clip_ratio/low_mean": 0.0331280380487442, "clip_ratio/low_min": 0.0331280380487442, "clip_ratio/high_mean": 0.08710999134927988, "clip_ratio/high_max": 0.08710999134927988, "clip_ratio/region_mean": 0.12023802939802408, "reward_total_mean": 0.5204696655273438, "reward_meter_mean": 0.7195549607276917, "reward_meter_std": 0.3614080846309662, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9357661008834839, "reward_repeat_soft_std": 0.04376447945833206, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.5204696655273438, "reward_total_composite_std": 0.13906559348106384} {"timestamp_utc": "2026-04-13T10:51:54Z", "mode": "train", "global_step": 1424, "epoch": 0.14304369663485686, "loss": 0.0055, "grad_norm": 8.282552719116211, "learning_rate": 5.687878787878789e-06, "num_tokens": 2511202.0, "completions/mean_length": 93.875, "completions/min_length": 87.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.875, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.6325433254241943, "rewards/meter/std": 0.3539314270019531, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8354396820068359, "rewards/repeat_soft/std": 0.05221690610051155, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.49850624799728394, "rewards/total_composite/std": 0.09172514826059341, "reward": 0.49850624799728394, "reward_std": 0.09172514081001282, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11522486060857773, "sampling/sampling_logp_difference/max": 2.3873190879821777, "sampling/importance_sampling_ratio/min": 0.09187566488981247, "sampling/importance_sampling_ratio/mean": 1.00412917137146, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6340465322136879, "clip_ratio/low_mean": 0.05321041587740183, "clip_ratio/low_min": 0.05321041587740183, "clip_ratio/high_mean": 0.055305938702076674, "clip_ratio/high_max": 0.055305938702076674, "clip_ratio/region_mean": 0.1085163545794785, "reward_total_mean": 0.49850624799728394, "reward_meter_mean": 0.6325433254241943, "reward_meter_std": 0.3539314270019531, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8354396820068359, "reward_repeat_soft_std": 0.05221690610051155, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.49850624799728394, "reward_total_composite_std": 0.09172514826059341} {"timestamp_utc": "2026-04-13T10:52:00Z", "mode": "train", "global_step": 1425, "epoch": 0.14314414866901054, "loss": 0.0471, "grad_norm": 9.898316383361816, "learning_rate": 5.684848484848485e-06, "num_tokens": 2513556.0, "completions/mean_length": 93.25, "completions/min_length": 87.0, "completions/max_length": 113.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.25, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 113.0, "rewards/meter/mean": 0.9084113240242004, "rewards/meter/std": 0.0786547064781189, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8716374039649963, "rewards/repeat_soft/std": 0.05684646591544151, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.6510645151138306, "rewards/total_composite/std": 0.13250632584095, "reward": 0.6510645151138306, "reward_std": 0.13250631093978882, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13459469377994537, "sampling/sampling_logp_difference/max": 1.9838438034057617, "sampling/importance_sampling_ratio/min": 0.1375395506620407, "sampling/importance_sampling_ratio/mean": 1.0100266933441162, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7431213334202766, "clip_ratio/low_mean": 0.07877854350954294, "clip_ratio/low_min": 0.07877854350954294, "clip_ratio/high_mean": 0.019875479396432638, "clip_ratio/high_max": 0.019875479396432638, "clip_ratio/region_mean": 0.09865402290597558, "reward_total_mean": 0.6510645151138306, "reward_meter_mean": 0.9084113240242004, "reward_meter_std": 0.0786547064781189, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8716374039649963, "reward_repeat_soft_std": 0.05684646591544151, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.6510645151138306, "reward_total_composite_std": 0.13250632584095} {"timestamp_utc": "2026-04-13T10:52:08Z", "mode": "train", "global_step": 1426, "epoch": 0.14324460070316425, "loss": 0.1294, "grad_norm": 8.25341796875, "learning_rate": 5.681818181818183e-06, "num_tokens": 2515217.0, "completions/mean_length": 58.625, "completions/min_length": 36.0, "completions/max_length": 179.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.625, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 179.0, "rewards/meter/mean": 0.6144726276397705, "rewards/meter/std": 0.373422235250473, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9629454612731934, "rewards/repeat_soft/std": 0.036584775894880295, "rewards/judge_quality/mean": 0.53125, "rewards/judge_quality/std": 0.21963852643966675, "rewards/total_composite/mean": 0.5260547399520874, "rewards/total_composite/std": 0.12817879021167755, "reward": 0.5260547399520874, "reward_std": 0.12817879021167755, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09847000986337662, "sampling/sampling_logp_difference/max": 1.1577715873718262, "sampling/importance_sampling_ratio/min": 0.31418555974960327, "sampling/importance_sampling_ratio/mean": 1.0045069456100464, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7123184055089951, "clip_ratio/low_mean": 0.07798398123122752, "clip_ratio/low_min": 0.07798398123122752, "clip_ratio/high_mean": 0.051209207624197006, "clip_ratio/high_max": 0.051209207624197006, "clip_ratio/region_mean": 0.12919318885542452, "reward_total_mean": 0.5260547399520874, "reward_meter_mean": 0.6144726276397705, "reward_meter_std": 0.373422235250473, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9629454612731934, "reward_repeat_soft_std": 0.036584775894880295, "reward_judge_quality_mean": 0.53125, "reward_judge_quality_std": 0.21963852643966675, "reward_total_composite_mean": 0.5260547399520874, "reward_total_composite_std": 0.12817879021167755} {"timestamp_utc": "2026-04-13T10:52:19Z", "mode": "train", "global_step": 1427, "epoch": 0.14334505273731793, "loss": -0.0972, "grad_norm": 3.392662525177002, "learning_rate": 5.67878787878788e-06, "num_tokens": 2516861.0, "completions/mean_length": 109.5, "completions/min_length": 44.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 52.000003814697266, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.4390566945075989, "rewards/meter/std": 0.35161668062210083, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9917404651641846, "rewards/repeat_soft/std": 0.006031770259141922, "rewards/judge_quality/mean": 0.45625001192092896, "rewards/judge_quality/std": 0.20982563495635986, "rewards/total_composite/mean": 0.45237046480178833, "rewards/total_composite/std": 0.22913609445095062, "reward": 0.45237046480178833, "reward_std": 0.22913609445095062, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1584887057542801, "sampling/sampling_logp_difference/max": 2.5816755294799805, "sampling/importance_sampling_ratio/min": 0.07564714550971985, "sampling/importance_sampling_ratio/mean": 1.0274124145507812, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.973654605448246, "clip_ratio/low_mean": 0.03229166753590107, "clip_ratio/low_min": 0.03229166753590107, "clip_ratio/high_mean": 0.07325901370495558, "clip_ratio/high_max": 0.07325901370495558, "clip_ratio/region_mean": 0.10555068124085665, "reward_total_mean": 0.45237046480178833, "reward_meter_mean": 0.4390566945075989, "reward_meter_std": 0.35161668062210083, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9917404651641846, "reward_repeat_soft_std": 0.006031770259141922, "reward_judge_quality_mean": 0.45625001192092896, "reward_judge_quality_std": 0.20982563495635986, "reward_total_composite_mean": 0.45237046480178833, "reward_total_composite_std": 0.22913609445095062} {"timestamp_utc": "2026-04-13T10:52:27Z", "mode": "train", "global_step": 1428, "epoch": 0.1434455047714716, "loss": 0.0098, "grad_norm": 10.655179023742676, "learning_rate": 5.675757575757577e-06, "num_tokens": 2518932.0, "completions/mean_length": 79.875, "completions/min_length": 55.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.875, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.5962883234024048, "rewards/meter/std": 0.327947735786438, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8625436425209045, "rewards/repeat_soft/std": 0.0839657112956047, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.5221672058105469, "rewards/total_composite/std": 0.13741447031497955, "reward": 0.5221672058105469, "reward_std": 0.13741447031497955, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1376197636127472, "sampling/sampling_logp_difference/max": 2.1869306564331055, "sampling/importance_sampling_ratio/min": 0.11226078122854233, "sampling/importance_sampling_ratio/mean": 1.0101064443588257, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7594892457127571, "clip_ratio/low_mean": 0.03768882201984525, "clip_ratio/low_min": 0.03768882201984525, "clip_ratio/high_mean": 0.06279271235689521, "clip_ratio/high_max": 0.06279271235689521, "clip_ratio/region_mean": 0.10048153437674046, "reward_total_mean": 0.5221672058105469, "reward_meter_mean": 0.5962883234024048, "reward_meter_std": 0.327947735786438, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8625436425209045, "reward_repeat_soft_std": 0.0839657112956047, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.5221672058105469, "reward_total_composite_std": 0.13741447031497955} {"timestamp_utc": "2026-04-13T10:52:33Z", "mode": "train", "global_step": 1429, "epoch": 0.14354595680562532, "loss": -0.0631, "grad_norm": 10.576534271240234, "learning_rate": 5.672727272727273e-06, "num_tokens": 2521051.0, "completions/mean_length": 76.875, "completions/min_length": 61.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.875, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.4533776044845581, "rewards/meter/std": 0.2749289572238922, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9741709232330322, "rewards/repeat_soft/std": 0.028394293040037155, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4698977470397949, "rewards/total_composite/std": 0.07737768441438675, "reward": 0.4698977470397949, "reward_std": 0.07737768441438675, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13355378806591034, "sampling/sampling_logp_difference/max": 1.7773722410202026, "sampling/importance_sampling_ratio/min": 0.1690818816423416, "sampling/importance_sampling_ratio/mean": 1.0167454481124878, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9134811013936996, "clip_ratio/low_mean": 0.05188595782965422, "clip_ratio/low_min": 0.05188595782965422, "clip_ratio/high_mean": 0.0636458182707429, "clip_ratio/high_max": 0.0636458182707429, "clip_ratio/region_mean": 0.11553177610039711, "reward_total_mean": 0.4698977470397949, "reward_meter_mean": 0.4533776044845581, "reward_meter_std": 0.2749289572238922, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9741709232330322, "reward_repeat_soft_std": 0.028394293040037155, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4698977470397949, "reward_total_composite_std": 0.07737768441438675} {"timestamp_utc": "2026-04-13T10:52:40Z", "mode": "train", "global_step": 1430, "epoch": 0.143646408839779, "loss": -0.0189, "grad_norm": 13.204424858093262, "learning_rate": 5.6696969696969705e-06, "num_tokens": 2522840.0, "completions/mean_length": 41.625, "completions/min_length": 37.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.625, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.8196427822113037, "rewards/meter/std": 0.31651848554611206, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9179916381835938, "rewards/repeat_soft/std": 0.0478409081697464, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.563774824142456, "rewards/total_composite/std": 0.08826941251754761, "reward": 0.563774824142456, "reward_std": 0.08826940506696701, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13667036592960358, "sampling/sampling_logp_difference/max": 1.5471887588500977, "sampling/importance_sampling_ratio/min": 0.21284548938274384, "sampling/importance_sampling_ratio/mean": 1.0013889074325562, "sampling/importance_sampling_ratio/max": 1.981207013130188, "entropy": 0.688811220228672, "clip_ratio/low_mean": 0.04380293842405081, "clip_ratio/low_min": 0.04380293842405081, "clip_ratio/high_mean": 0.061815458349883556, "clip_ratio/high_max": 0.061815458349883556, "clip_ratio/region_mean": 0.10561839677393436, "reward_total_mean": 0.563774824142456, "reward_meter_mean": 0.8196427822113037, "reward_meter_std": 0.31651848554611206, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9179916381835938, "reward_repeat_soft_std": 0.0478409081697464, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.563774824142456, "reward_total_composite_std": 0.08826941251754761} {"timestamp_utc": "2026-04-13T10:52:46Z", "mode": "train", "global_step": 1431, "epoch": 0.1437468608739327, "loss": 0.0257, "grad_norm": 12.921917915344238, "learning_rate": 5.666666666666667e-06, "num_tokens": 2524347.0, "completions/mean_length": 40.375, "completions/min_length": 36.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.375, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.954035758972168, "rewards/meter/std": 0.034081488847732544, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9922205209732056, "rewards/repeat_soft/std": 0.008540960028767586, "rewards/judge_quality/mean": 0.581250011920929, "rewards/judge_quality/std": 0.22183892130851746, "rewards/total_composite/mean": 0.7052081227302551, "rewards/total_composite/std": 0.12278620153665543, "reward": 0.7052081227302551, "reward_std": 0.12278620153665543, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12639567255973816, "sampling/sampling_logp_difference/max": 1.8599350452423096, "sampling/importance_sampling_ratio/min": 0.1556827425956726, "sampling/importance_sampling_ratio/mean": 0.9986779093742371, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7381427623331547, "clip_ratio/low_mean": 0.07170370407402515, "clip_ratio/low_min": 0.07170370407402515, "clip_ratio/high_mean": 0.05122265079990029, "clip_ratio/high_max": 0.05122265079990029, "clip_ratio/region_mean": 0.12292635487392545, "reward_total_mean": 0.7052081227302551, "reward_meter_mean": 0.954035758972168, "reward_meter_std": 0.034081488847732544, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9922205209732056, "reward_repeat_soft_std": 0.008540960028767586, "reward_judge_quality_mean": 0.581250011920929, "reward_judge_quality_std": 0.22183892130851746, "reward_total_composite_mean": 0.7052081227302551, "reward_total_composite_std": 0.12278620153665543} {"timestamp_utc": "2026-04-13T10:52:57Z", "mode": "train", "global_step": 1432, "epoch": 0.1438473129080864, "loss": -0.1825, "grad_norm": 2.517871141433716, "learning_rate": 5.663636363636364e-06, "num_tokens": 2526584.0, "completions/mean_length": 145.625, "completions/min_length": 80.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 93.28572082519531, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.66487056016922, "rewards/meter/std": 0.30841493606567383, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8946514129638672, "rewards/repeat_soft/std": 0.06716026365756989, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.4258626103401184, "rewards/total_composite/std": 0.1878291219472885, "reward": 0.4258626103401184, "reward_std": 0.1878291219472885, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13334038853645325, "sampling/sampling_logp_difference/max": 3.3211793899536133, "sampling/importance_sampling_ratio/min": 0.03611021861433983, "sampling/importance_sampling_ratio/mean": 1.012913465499878, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5987378433346748, "clip_ratio/low_mean": 0.028645833022892475, "clip_ratio/low_min": 0.028645833022892475, "clip_ratio/high_mean": 0.06977557484060526, "clip_ratio/high_max": 0.06977557484060526, "clip_ratio/region_mean": 0.09842140786349773, "reward_total_mean": 0.4258626103401184, "reward_meter_mean": 0.66487056016922, "reward_meter_std": 0.30841493606567383, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8946514129638672, "reward_repeat_soft_std": 0.06716026365756989, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.4258626103401184, "reward_total_composite_std": 0.1878291219472885} {"timestamp_utc": "2026-04-13T10:53:03Z", "mode": "train", "global_step": 1433, "epoch": 0.14394776494224007, "loss": 0.0887, "grad_norm": 12.529499053955078, "learning_rate": 5.6606060606060606e-06, "num_tokens": 2528272.0, "completions/mean_length": 50.0, "completions/min_length": 42.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.0, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.6653802394866943, "rewards/meter/std": 0.3758772611618042, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9062150716781616, "rewards/repeat_soft/std": 0.12127143889665604, "rewards/judge_quality/mean": 0.6862500309944153, "rewards/judge_quality/std": 0.27406659722328186, "rewards/total_composite/mean": 0.6130000352859497, "rewards/total_composite/std": 0.20596526563167572, "reward": 0.6130000352859497, "reward_std": 0.20596528053283691, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1393696516752243, "sampling/sampling_logp_difference/max": 1.537881851196289, "sampling/importance_sampling_ratio/min": 0.21483567357063293, "sampling/importance_sampling_ratio/mean": 0.9889280200004578, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.724241353571415, "clip_ratio/low_mean": 0.05138481268659234, "clip_ratio/low_min": 0.05138481268659234, "clip_ratio/high_mean": 0.07794269174337387, "clip_ratio/high_max": 0.07794269174337387, "clip_ratio/region_mean": 0.1293275044299662, "reward_total_mean": 0.6130000352859497, "reward_meter_mean": 0.6653802394866943, "reward_meter_std": 0.3758772611618042, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9062150716781616, "reward_repeat_soft_std": 0.12127143889665604, "reward_judge_quality_mean": 0.6862500309944153, "reward_judge_quality_std": 0.27406659722328186, "reward_total_composite_mean": 0.6130000352859497, "reward_total_composite_std": 0.20596526563167572} {"timestamp_utc": "2026-04-13T10:53:10Z", "mode": "train", "global_step": 1434, "epoch": 0.14404821697639378, "loss": -0.0152, "grad_norm": 12.750229835510254, "learning_rate": 5.657575757575759e-06, "num_tokens": 2529611.0, "completions/mean_length": 43.375, "completions/min_length": 36.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.375, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.2933866083621979, "rewards/meter/std": 0.313712477684021, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9353867769241333, "rewards/repeat_soft/std": 0.08880577236413956, "rewards/judge_quality/mean": 0.47749999165534973, "rewards/judge_quality/std": 0.16184209287166595, "rewards/total_composite/mean": 0.42346441745758057, "rewards/total_composite/std": 0.0754440650343895, "reward": 0.42346441745758057, "reward_std": 0.0754440575838089, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1647147387266159, "sampling/sampling_logp_difference/max": 1.2110676765441895, "sampling/importance_sampling_ratio/min": 0.29787907004356384, "sampling/importance_sampling_ratio/mean": 1.0240682363510132, "sampling/importance_sampling_ratio/max": 1.938675880432129, "entropy": 1.508640632033348, "clip_ratio/low_mean": 0.07083119731396437, "clip_ratio/low_min": 0.07083119731396437, "clip_ratio/high_mean": 0.05808390025049448, "clip_ratio/high_max": 0.05808390025049448, "clip_ratio/region_mean": 0.12891509756445885, "reward_total_mean": 0.42346441745758057, "reward_meter_mean": 0.2933866083621979, "reward_meter_std": 0.313712477684021, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9353867769241333, "reward_repeat_soft_std": 0.08880577236413956, "reward_judge_quality_mean": 0.47749999165534973, "reward_judge_quality_std": 0.16184209287166595, "reward_total_composite_mean": 0.42346441745758057, "reward_total_composite_std": 0.0754440650343895} {"timestamp_utc": "2026-04-13T10:53:16Z", "mode": "train", "global_step": 1435, "epoch": 0.14414866901054746, "loss": 0.0928, "grad_norm": 17.791757583618164, "learning_rate": 5.654545454545455e-06, "num_tokens": 2531193.0, "completions/mean_length": 38.75, "completions/min_length": 34.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.75, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.6358752250671387, "rewards/meter/std": 0.29427236318588257, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9202096462249756, "rewards/repeat_soft/std": 0.05021527409553528, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.49828529357910156, "rewards/total_composite/std": 0.07509937137365341, "reward": 0.49828529357910156, "reward_std": 0.07509937137365341, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1439976990222931, "sampling/sampling_logp_difference/max": 1.0636416673660278, "sampling/importance_sampling_ratio/min": 0.3451964259147644, "sampling/importance_sampling_ratio/mean": 1.0063138008117676, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7885511443018913, "clip_ratio/low_mean": 0.07318278471939266, "clip_ratio/low_min": 0.07318278471939266, "clip_ratio/high_mean": 0.06283324770629406, "clip_ratio/high_max": 0.06283324770629406, "clip_ratio/region_mean": 0.13601603242568672, "reward_total_mean": 0.49828529357910156, "reward_meter_mean": 0.6358752250671387, "reward_meter_std": 0.29427236318588257, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9202096462249756, "reward_repeat_soft_std": 0.05021527409553528, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.49828529357910156, "reward_total_composite_std": 0.07509937137365341} {"timestamp_utc": "2026-04-13T10:53:21Z", "mode": "train", "global_step": 1436, "epoch": 0.14424912104470117, "loss": 0.0895, "grad_norm": 17.731231689453125, "learning_rate": 5.651515151515152e-06, "num_tokens": 2532600.0, "completions/mean_length": 19.875, "completions/min_length": 17.0, "completions/max_length": 23.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 19.875, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 23.0, "rewards/meter/mean": 0.8622947335243225, "rewards/meter/std": 0.32788988947868347, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9495408535003662, "rewards/repeat_soft/std": 0.024242952466011047, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5778375864028931, "rewards/total_composite/std": 0.0893905982375145, "reward": 0.5778375864028931, "reward_std": 0.0893905833363533, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10792140662670135, "sampling/sampling_logp_difference/max": 1.2819077968597412, "sampling/importance_sampling_ratio/min": 0.27750736474990845, "sampling/importance_sampling_ratio/mean": 0.998196542263031, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.515293225646019, "clip_ratio/low_mean": 0.010869565419852734, "clip_ratio/low_min": 0.010869565419852734, "clip_ratio/high_mean": 0.09196286275982857, "clip_ratio/high_max": 0.09196286275982857, "clip_ratio/region_mean": 0.1028324281796813, "reward_total_mean": 0.5778375864028931, "reward_meter_mean": 0.8622947335243225, "reward_meter_std": 0.32788988947868347, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9495408535003662, "reward_repeat_soft_std": 0.024242952466011047, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5778375864028931, "reward_total_composite_std": 0.0893905982375145} {"timestamp_utc": "2026-04-13T10:53:28Z", "mode": "train", "global_step": 1437, "epoch": 0.14434957307885485, "loss": 0.0494, "grad_norm": 12.390629768371582, "learning_rate": 5.648484848484849e-06, "num_tokens": 2534172.0, "completions/mean_length": 38.5, "completions/min_length": 36.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.5, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.7535187005996704, "rewards/meter/std": 0.3593631088733673, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9046169519424438, "rewards/repeat_soft/std": 0.07597479969263077, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.631027340888977, "rewards/total_composite/std": 0.17327415943145752, "reward": 0.631027340888977, "reward_std": 0.17327415943145752, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1142137199640274, "sampling/sampling_logp_difference/max": 2.5830631256103516, "sampling/importance_sampling_ratio/min": 0.07554225623607635, "sampling/importance_sampling_ratio/mean": 1.008995771408081, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6440048366785049, "clip_ratio/low_mean": 0.07128219399601221, "clip_ratio/low_min": 0.07128219399601221, "clip_ratio/high_mean": 0.032894738018512726, "clip_ratio/high_max": 0.032894738018512726, "clip_ratio/region_mean": 0.10417693201452494, "reward_total_mean": 0.631027340888977, "reward_meter_mean": 0.7535187005996704, "reward_meter_std": 0.3593631088733673, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9046169519424438, "reward_repeat_soft_std": 0.07597479969263077, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.631027340888977, "reward_total_composite_std": 0.17327415943145752} {"timestamp_utc": "2026-04-13T10:53:34Z", "mode": "train", "global_step": 1438, "epoch": 0.14445002511300853, "loss": 0.0082, "grad_norm": 8.533262252807617, "learning_rate": 5.645454545454546e-06, "num_tokens": 2536644.0, "completions/mean_length": 93.0, "completions/min_length": 84.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.0, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.6551315784454346, "rewards/meter/std": 0.4282614588737488, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.05892555043101311, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7912271618843079, "rewards/repeat_soft/std": 0.07064896076917648, "rewards/judge_quality/mean": 0.47749996185302734, "rewards/judge_quality/std": 0.16263456642627716, "rewards/total_composite/mean": 0.4958537817001343, "rewards/total_composite/std": 0.1731943041086197, "reward": 0.4958537817001343, "reward_std": 0.1731942892074585, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1107155978679657, "sampling/sampling_logp_difference/max": 1.740966558456421, "sampling/importance_sampling_ratio/min": 0.1753508448600769, "sampling/importance_sampling_ratio/mean": 1.0029927492141724, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5855625607073307, "clip_ratio/low_mean": 0.039472698234021664, "clip_ratio/low_min": 0.039472698234021664, "clip_ratio/high_mean": 0.06870181625708938, "clip_ratio/high_max": 0.06870181625708938, "clip_ratio/region_mean": 0.10817451449111104, "reward_total_mean": 0.4958537817001343, "reward_meter_mean": 0.6551315784454346, "reward_meter_std": 0.4282614588737488, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.05892555043101311, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7912271618843079, "reward_repeat_soft_std": 0.07064896076917648, "reward_judge_quality_mean": 0.47749996185302734, "reward_judge_quality_std": 0.16263456642627716, "reward_total_composite_mean": 0.4958537817001343, "reward_total_composite_std": 0.1731943041086197} {"timestamp_utc": "2026-04-13T10:53:41Z", "mode": "train", "global_step": 1439, "epoch": 0.14455047714716224, "loss": 0.0215, "grad_norm": 10.35162353515625, "learning_rate": 5.642424242424242e-06, "num_tokens": 2538462.0, "completions/mean_length": 65.25, "completions/min_length": 61.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.25, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.3177836239337921, "rewards/meter/std": 0.27080944180488586, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8657203316688538, "rewards/repeat_soft/std": 0.06126207858324051, "rewards/judge_quality/mean": 0.6449999809265137, "rewards/judge_quality/std": 0.24928471446037292, "rewards/total_composite/mean": 0.4653521776199341, "rewards/total_composite/std": 0.15516647696495056, "reward": 0.4653521776199341, "reward_std": 0.15516646206378937, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11212596297264099, "sampling/sampling_logp_difference/max": 1.9287055730819702, "sampling/importance_sampling_ratio/min": 0.14533619582653046, "sampling/importance_sampling_ratio/mean": 1.0013866424560547, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.36079560965299606, "clip_ratio/low_mean": 0.07574523240327835, "clip_ratio/low_min": 0.07574523240327835, "clip_ratio/high_mean": 0.03560520429164171, "clip_ratio/high_max": 0.03560520429164171, "clip_ratio/region_mean": 0.11135043669492006, "reward_total_mean": 0.4653521776199341, "reward_meter_mean": 0.3177836239337921, "reward_meter_std": 0.27080944180488586, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8657203316688538, "reward_repeat_soft_std": 0.06126207858324051, "reward_judge_quality_mean": 0.6449999809265137, "reward_judge_quality_std": 0.24928471446037292, "reward_total_composite_mean": 0.4653521776199341, "reward_total_composite_std": 0.15516647696495056} {"timestamp_utc": "2026-04-13T10:53:49Z", "mode": "train", "global_step": 1440, "epoch": 0.14465092918131592, "loss": 0.0367, "grad_norm": 7.798024654388428, "learning_rate": 5.6393939393939405e-06, "num_tokens": 2540855.0, "completions/mean_length": 94.125, "completions/min_length": 81.0, "completions/max_length": 118.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.125, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 118.0, "rewards/meter/mean": 0.09384974837303162, "rewards/meter/std": 0.07115980982780457, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8989412784576416, "rewards/repeat_soft/std": 0.0413278266787529, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.32009556889533997, "rewards/total_composite/std": 0.019734904170036316, "reward": 0.32009556889533997, "reward_std": 0.019734907895326614, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12110128998756409, "sampling/sampling_logp_difference/max": 3.317157745361328, "sampling/importance_sampling_ratio/min": 0.03625573590397835, "sampling/importance_sampling_ratio/mean": 1.006064534187317, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6290554516017437, "clip_ratio/low_mean": 0.07291316520422697, "clip_ratio/low_min": 0.07291316520422697, "clip_ratio/high_mean": 0.03113451600074768, "clip_ratio/high_max": 0.03113451600074768, "clip_ratio/region_mean": 0.10404768120497465, "reward_total_mean": 0.32009556889533997, "reward_meter_mean": 0.09384974837303162, "reward_meter_std": 0.07115980982780457, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8989412784576416, "reward_repeat_soft_std": 0.0413278266787529, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.32009556889533997, "reward_total_composite_std": 0.019734904170036316} {"timestamp_utc": "2026-04-13T10:54:00Z", "mode": "train", "global_step": 1441, "epoch": 0.14475138121546963, "loss": -0.0692, "grad_norm": 3.9723446369171143, "learning_rate": 5.636363636363636e-06, "num_tokens": 2542352.0, "completions/mean_length": 93.125, "completions/min_length": 26.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 33.28571701049805, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.3827114999294281, "rewards/meter/std": 0.3394012451171875, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9972173571586609, "rewards/repeat_soft/std": 0.005236493889242411, "rewards/judge_quality/mean": 0.6187499761581421, "rewards/judge_quality/std": 0.33434104919433594, "rewards/total_composite/mean": 0.48450589179992676, "rewards/total_composite/std": 0.2517508268356323, "reward": 0.48450589179992676, "reward_std": 0.2517508268356323, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12247990816831589, "sampling/sampling_logp_difference/max": 1.6840438842773438, "sampling/importance_sampling_ratio/min": 0.18562181293964386, "sampling/importance_sampling_ratio/mean": 1.0007109642028809, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3911283649504185, "clip_ratio/low_mean": 0.047941374592483044, "clip_ratio/low_min": 0.047941374592483044, "clip_ratio/high_mean": 0.04840888362377882, "clip_ratio/high_max": 0.04840888362377882, "clip_ratio/region_mean": 0.09635025821626186, "reward_total_mean": 0.48450589179992676, "reward_meter_mean": 0.3827114999294281, "reward_meter_std": 0.3394012451171875, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9972173571586609, "reward_repeat_soft_std": 0.005236493889242411, "reward_judge_quality_mean": 0.6187499761581421, "reward_judge_quality_std": 0.33434104919433594, "reward_total_composite_mean": 0.48450589179992676, "reward_total_composite_std": 0.2517508268356323} {"timestamp_utc": "2026-04-13T10:54:06Z", "mode": "train", "global_step": 1442, "epoch": 0.1448518332496233, "loss": 0.1372, "grad_norm": 12.586616516113281, "learning_rate": 5.633333333333334e-06, "num_tokens": 2543784.0, "completions/mean_length": 40.0, "completions/min_length": 33.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.0, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.4086367189884186, "rewards/meter/std": 0.3910621404647827, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9808080196380615, "rewards/repeat_soft/std": 0.021186163648962975, "rewards/judge_quality/mean": 0.5262500047683716, "rewards/judge_quality/std": 0.17079123854637146, "rewards/total_composite/mean": 0.4698314368724823, "rewards/total_composite/std": 0.11303868889808655, "reward": 0.4698314368724823, "reward_std": 0.11303868889808655, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14552311599254608, "sampling/sampling_logp_difference/max": 1.7716580629348755, "sampling/importance_sampling_ratio/min": 0.17005079984664917, "sampling/importance_sampling_ratio/mean": 1.0104730129241943, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7147787809371948, "clip_ratio/low_mean": 0.08094080351293087, "clip_ratio/low_min": 0.08094080351293087, "clip_ratio/high_mean": 0.056038325652480125, "clip_ratio/high_max": 0.056038325652480125, "clip_ratio/region_mean": 0.136979129165411, "reward_total_mean": 0.4698314368724823, "reward_meter_mean": 0.4086367189884186, "reward_meter_std": 0.3910621404647827, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9808080196380615, "reward_repeat_soft_std": 0.021186163648962975, "reward_judge_quality_mean": 0.5262500047683716, "reward_judge_quality_std": 0.17079123854637146, "reward_total_composite_mean": 0.4698314368724823, "reward_total_composite_std": 0.11303868889808655} {"timestamp_utc": "2026-04-13T10:54:17Z", "mode": "train", "global_step": 1443, "epoch": 0.144952285283777, "loss": -0.1288, "grad_norm": 4.243305206298828, "learning_rate": 5.630303030303031e-06, "num_tokens": 2545698.0, "completions/mean_length": 120.25, "completions/min_length": 53.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 64.28572082519531, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.6812485456466675, "rewards/meter/std": 0.2924301028251648, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9527044296264648, "rewards/repeat_soft/std": 0.013067924417555332, "rewards/judge_quality/mean": 0.48625001311302185, "rewards/judge_quality/std": 0.23046152293682098, "rewards/total_composite/mean": 0.5220375657081604, "rewards/total_composite/std": 0.2517196238040924, "reward": 0.5220375657081604, "reward_std": 0.25171959400177, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16307465732097626, "sampling/sampling_logp_difference/max": 1.9829330444335938, "sampling/importance_sampling_ratio/min": 0.13766486942768097, "sampling/importance_sampling_ratio/mean": 0.9993640184402466, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7299496084451675, "clip_ratio/low_mean": 0.036057693883776665, "clip_ratio/low_min": 0.036057693883776665, "clip_ratio/high_mean": 0.11155798844993114, "clip_ratio/high_max": 0.11155798844993114, "clip_ratio/region_mean": 0.1476156823337078, "reward_total_mean": 0.5220375657081604, "reward_meter_mean": 0.6812485456466675, "reward_meter_std": 0.2924301028251648, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9527044296264648, "reward_repeat_soft_std": 0.013067924417555332, "reward_judge_quality_mean": 0.48625001311302185, "reward_judge_quality_std": 0.23046152293682098, "reward_total_composite_mean": 0.5220375657081604, "reward_total_composite_std": 0.2517196238040924} {"timestamp_utc": "2026-04-13T10:54:24Z", "mode": "train", "global_step": 1444, "epoch": 0.1450527373179307, "loss": 0.0471, "grad_norm": 6.888050079345703, "learning_rate": 5.627272727272728e-06, "num_tokens": 2548026.0, "completions/mean_length": 92.0, "completions/min_length": 78.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.0, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.4131786823272705, "rewards/meter/std": 0.24545283615589142, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8094701766967773, "rewards/repeat_soft/std": 0.05194660276174545, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.3885853886604309, "rewards/total_composite/std": 0.05554511770606041, "reward": 0.3885853886604309, "reward_std": 0.05554512143135071, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10161849856376648, "sampling/sampling_logp_difference/max": 2.4903271198272705, "sampling/importance_sampling_ratio/min": 0.0828828513622284, "sampling/importance_sampling_ratio/mean": 1.0064361095428467, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6244943141937256, "clip_ratio/low_mean": 0.03710805019363761, "clip_ratio/low_min": 0.03710805019363761, "clip_ratio/high_mean": 0.04541767109185457, "clip_ratio/high_max": 0.04541767109185457, "clip_ratio/region_mean": 0.08252572128549218, "reward_total_mean": 0.3885853886604309, "reward_meter_mean": 0.4131786823272705, "reward_meter_std": 0.24545283615589142, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8094701766967773, "reward_repeat_soft_std": 0.05194660276174545, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.3885853886604309, "reward_total_composite_std": 0.05554511770606041} {"timestamp_utc": "2026-04-13T10:54:31Z", "mode": "train", "global_step": 1445, "epoch": 0.14515318935208438, "loss": 0.1322, "grad_norm": 16.269306182861328, "learning_rate": 5.624242424242424e-06, "num_tokens": 2549506.0, "completions/mean_length": 26.0, "completions/min_length": 23.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.0, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.43530893325805664, "rewards/meter/std": 0.35694068670272827, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9530594348907471, "rewards/repeat_soft/std": 0.026701925322413445, "rewards/judge_quality/mean": 0.4437499940395355, "rewards/judge_quality/std": 0.2084594964981079, "rewards/total_composite/mean": 0.4565836191177368, "rewards/total_composite/std": 0.09808404743671417, "reward": 0.4565836191177368, "reward_std": 0.09808403998613358, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14200277626514435, "sampling/sampling_logp_difference/max": 1.2099218368530273, "sampling/importance_sampling_ratio/min": 0.29822060465812683, "sampling/importance_sampling_ratio/mean": 1.0090938806533813, "sampling/importance_sampling_ratio/max": 1.6397645473480225, "entropy": 0.9584878236055374, "clip_ratio/low_mean": 0.04872150160372257, "clip_ratio/low_min": 0.04872150160372257, "clip_ratio/high_mean": 0.05713768117129803, "clip_ratio/high_max": 0.05713768117129803, "clip_ratio/region_mean": 0.1058591827750206, "reward_total_mean": 0.4565836191177368, "reward_meter_mean": 0.43530893325805664, "reward_meter_std": 0.35694068670272827, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9530594348907471, "reward_repeat_soft_std": 0.026701925322413445, "reward_judge_quality_mean": 0.4437499940395355, "reward_judge_quality_std": 0.2084594964981079, "reward_total_composite_mean": 0.4565836191177368, "reward_total_composite_std": 0.09808404743671417} {"timestamp_utc": "2026-04-13T10:54:36Z", "mode": "train", "global_step": 1446, "epoch": 0.14525364138623806, "loss": -0.0025, "grad_norm": 12.225137710571289, "learning_rate": 5.6212121212121215e-06, "num_tokens": 2551060.0, "completions/mean_length": 32.25, "completions/min_length": 28.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.25, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.9132502675056458, "rewards/meter/std": 0.09246927499771118, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9432229995727539, "rewards/repeat_soft/std": 0.05181838199496269, "rewards/judge_quality/mean": 0.4137499928474426, "rewards/judge_quality/std": 0.06781013309955597, "rewards/total_composite/mean": 0.5862990617752075, "rewards/total_composite/std": 0.04857126250863075, "reward": 0.5862990617752075, "reward_std": 0.04857126623392105, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09656811505556107, "sampling/sampling_logp_difference/max": 1.240281581878662, "sampling/importance_sampling_ratio/min": 0.2893027663230896, "sampling/importance_sampling_ratio/mean": 1.0012781620025635, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5067105516791344, "clip_ratio/low_mean": 0.04288594517856836, "clip_ratio/low_min": 0.04288594517856836, "clip_ratio/high_mean": 0.04291552724316716, "clip_ratio/high_max": 0.04291552724316716, "clip_ratio/region_mean": 0.08580147242173553, "reward_total_mean": 0.5862990617752075, "reward_meter_mean": 0.9132502675056458, "reward_meter_std": 0.09246927499771118, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9432229995727539, "reward_repeat_soft_std": 0.05181838199496269, "reward_judge_quality_mean": 0.4137499928474426, "reward_judge_quality_std": 0.06781013309955597, "reward_total_composite_mean": 0.5862990617752075, "reward_total_composite_std": 0.04857126250863075} {"timestamp_utc": "2026-04-13T10:54:42Z", "mode": "train", "global_step": 1447, "epoch": 0.14535409342039177, "loss": 0.068, "grad_norm": 13.407052040100098, "learning_rate": 5.618181818181818e-06, "num_tokens": 2552642.0, "completions/mean_length": 37.75, "completions/min_length": 32.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.75, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.4962160289287567, "rewards/meter/std": 0.3201512396335602, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9090368747711182, "rewards/repeat_soft/std": 0.04927627369761467, "rewards/judge_quality/mean": 0.7112500071525574, "rewards/judge_quality/std": 0.24793073534965515, "rewards/total_composite/mean": 0.5681751370429993, "rewards/total_composite/std": 0.17873920500278473, "reward": 0.5681751370429993, "reward_std": 0.17873920500278473, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.118393175303936, "sampling/sampling_logp_difference/max": 1.5386567115783691, "sampling/importance_sampling_ratio/min": 0.21466928720474243, "sampling/importance_sampling_ratio/mean": 0.9922995567321777, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6725741475820541, "clip_ratio/low_mean": 0.05689032329246402, "clip_ratio/low_min": 0.05689032329246402, "clip_ratio/high_mean": 0.04010617733001709, "clip_ratio/high_max": 0.04010617733001709, "clip_ratio/region_mean": 0.09699650062248111, "reward_total_mean": 0.5681751370429993, "reward_meter_mean": 0.4962160289287567, "reward_meter_std": 0.3201512396335602, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9090368747711182, "reward_repeat_soft_std": 0.04927627369761467, "reward_judge_quality_mean": 0.7112500071525574, "reward_judge_quality_std": 0.24793073534965515, "reward_total_composite_mean": 0.5681751370429993, "reward_total_composite_std": 0.17873920500278473} {"timestamp_utc": "2026-04-13T10:54:48Z", "mode": "train", "global_step": 1448, "epoch": 0.14545454545454545, "loss": 0.0022, "grad_norm": 18.25330352783203, "learning_rate": 5.615151515151516e-06, "num_tokens": 2554090.0, "completions/mean_length": 34.0, "completions/min_length": 31.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.0, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.7931654453277588, "rewards/meter/std": 0.2793334424495697, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9400959014892578, "rewards/repeat_soft/std": 0.05659913271665573, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.5964663624763489, "rewards/total_composite/std": 0.14883674681186676, "reward": 0.5964663624763489, "reward_std": 0.14883676171302795, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10333961993455887, "sampling/sampling_logp_difference/max": 1.230234146118164, "sampling/importance_sampling_ratio/min": 0.29222413897514343, "sampling/importance_sampling_ratio/mean": 0.998220682144165, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5057181008160114, "clip_ratio/low_mean": 0.0633125938475132, "clip_ratio/low_min": 0.0633125938475132, "clip_ratio/high_mean": 0.04446381703019142, "clip_ratio/high_max": 0.04446381703019142, "clip_ratio/region_mean": 0.10777641087770462, "reward_total_mean": 0.5964663624763489, "reward_meter_mean": 0.7931654453277588, "reward_meter_std": 0.2793334424495697, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9400959014892578, "reward_repeat_soft_std": 0.05659913271665573, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.5964663624763489, "reward_total_composite_std": 0.14883674681186676} {"timestamp_utc": "2026-04-13T10:54:55Z", "mode": "train", "global_step": 1449, "epoch": 0.14555499748869916, "loss": -0.0235, "grad_norm": 11.540170669555664, "learning_rate": 5.612121212121212e-06, "num_tokens": 2556021.0, "completions/mean_length": 67.375, "completions/min_length": 55.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.375, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.9395323991775513, "rewards/meter/std": 0.05084110423922539, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8415687084197998, "rewards/repeat_soft/std": 0.03868628293275833, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.5630892515182495, "rewards/total_composite/std": 0.04056117311120033, "reward": 0.5630892515182495, "reward_std": 0.04056116193532944, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08219168335199356, "sampling/sampling_logp_difference/max": 2.3244993686676025, "sampling/importance_sampling_ratio/min": 0.09783240407705307, "sampling/importance_sampling_ratio/mean": 1.0111548900604248, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.37682488933205605, "clip_ratio/low_mean": 0.03030737955123186, "clip_ratio/low_min": 0.03030737955123186, "clip_ratio/high_mean": 0.04899826226755977, "clip_ratio/high_max": 0.04899826226755977, "clip_ratio/region_mean": 0.07930564181879163, "reward_total_mean": 0.5630892515182495, "reward_meter_mean": 0.9395323991775513, "reward_meter_std": 0.05084110423922539, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8415687084197998, "reward_repeat_soft_std": 0.03868628293275833, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.5630892515182495, "reward_total_composite_std": 0.04056117311120033} {"timestamp_utc": "2026-04-13T10:55:02Z", "mode": "train", "global_step": 1450, "epoch": 0.14565544952285284, "loss": 0.0188, "grad_norm": 11.66684627532959, "learning_rate": 5.60909090909091e-06, "num_tokens": 2557654.0, "completions/mean_length": 42.125, "completions/min_length": 36.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.125, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.3843640089035034, "rewards/meter/std": 0.3152768909931183, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9392750263214111, "rewards/repeat_soft/std": 0.03424127772450447, "rewards/judge_quality/mean": 0.4424999952316284, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.45239052176475525, "rewards/total_composite/std": 0.09135284274816513, "reward": 0.45239052176475525, "reward_std": 0.09135284274816513, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12498661130666733, "sampling/sampling_logp_difference/max": 2.2886857986450195, "sampling/importance_sampling_ratio/min": 0.10139963775873184, "sampling/importance_sampling_ratio/mean": 1.0153913497924805, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7969678267836571, "clip_ratio/low_mean": 0.09065534360706806, "clip_ratio/low_min": 0.09065534360706806, "clip_ratio/high_mean": 0.03164640720933676, "clip_ratio/high_max": 0.03164640720933676, "clip_ratio/region_mean": 0.12230175081640482, "reward_total_mean": 0.45239052176475525, "reward_meter_mean": 0.3843640089035034, "reward_meter_std": 0.3152768909931183, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9392750263214111, "reward_repeat_soft_std": 0.03424127772450447, "reward_judge_quality_mean": 0.4424999952316284, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.45239052176475525, "reward_total_composite_std": 0.09135284274816513} {"timestamp_utc": "2026-04-13T10:55:40Z", "mode": "eval", "global_step": 1450, "epoch": 0.14565544952285284, "eval_loss": NaN, "eval_runtime": 37.6201, "eval_samples_per_second": 2.127, "eval_steps_per_second": 0.266, "eval_num_tokens": 2557654.0, "eval_completions/mean_length": 65.45, "eval_completions/min_length": 30.7, "eval_completions/max_length": 103.8, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 65.45, "eval_completions/min_terminated_length": 30.7, "eval_completions/max_terminated_length": 103.8, "eval_rewards/meter/mean": 0.5202593326568603, "eval_rewards/meter/std": 0.3611732065677643, "eval_rewards/count_adherence/mean": 0.9629166543483734, "eval_rewards/count_adherence/std": 0.06404209546744824, "eval_rewards/hard_gate/mean": 1.0, "eval_rewards/hard_gate/std": 0.0, "eval_rewards/repeat_soft/mean": 0.9235949277877807, "eval_rewards/repeat_soft/std": 0.06313590295612811, "eval_rewards/judge_quality/mean": 0.4716249942779541, "eval_rewards/judge_quality/std": 0.1344312877394259, "eval_rewards/total_composite/mean": 0.4841545969247818, "eval_rewards/total_composite/std": 0.11240602880716324, "eval_reward": 0.4841545969247818, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.06021547056734562, "eval_sampling/sampling_logp_difference/max": 1.0568527698516845, "eval_sampling/importance_sampling_ratio/min": 0.354967337846756, "eval_sampling/importance_sampling_ratio/mean": 1.0121062040328979, "eval_sampling/importance_sampling_ratio/max": 1.403704333305359, "eval_entropy": 0.6505220174789429, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.4841545969247818, "eval_reward_meter_mean": 0.5202593326568603, "eval_reward_meter_std": 0.3611732065677643, "eval_reward_count_adherence_mean": 0.9629166543483734, "eval_reward_count_adherence_std": 0.06404209546744824, "eval_reward_hard_gate_mean": 1.0, "eval_reward_hard_gate_std": 0.0, "eval_reward_repeat_soft_mean": 0.9235949277877807, "eval_reward_repeat_soft_std": 0.06313590295612811, "eval_reward_judge_quality_mean": 0.4716249942779541, "eval_reward_judge_quality_std": 0.1344312877394259, "eval_reward_total_composite_mean": 0.4841545969247818, "eval_reward_total_composite_std": 0.11240602880716324} {"timestamp_utc": "2026-04-13T10:55:48Z", "mode": "train", "global_step": 1451, "epoch": 0.14575590155700652, "loss": -0.0034, "grad_norm": 16.953227996826172, "learning_rate": 5.606060606060606e-06, "num_tokens": 2559106.0, "completions/mean_length": 28.5, "completions/min_length": 26.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.5, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.6778678297996521, "rewards/meter/std": 0.3919105529785156, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9157227277755737, "rewards/repeat_soft/std": 0.05926491692662239, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.6333736777305603, "rewards/total_composite/std": 0.22816839814186096, "reward": 0.6333736777305603, "reward_std": 0.22816839814186096, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11051974445581436, "sampling/sampling_logp_difference/max": 1.0624847412109375, "sampling/importance_sampling_ratio/min": 0.3455960154533386, "sampling/importance_sampling_ratio/mean": 1.0067046880722046, "sampling/importance_sampling_ratio/max": 1.7910748720169067, "entropy": 0.6300291679799557, "clip_ratio/low_mean": 0.06631200667470694, "clip_ratio/low_min": 0.06631200667470694, "clip_ratio/high_mean": 0.0210628192871809, "clip_ratio/high_max": 0.0210628192871809, "clip_ratio/region_mean": 0.08737482596188784, "reward_total_mean": 0.6333736777305603, "reward_meter_mean": 0.6778678297996521, "reward_meter_std": 0.3919105529785156, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9157227277755737, "reward_repeat_soft_std": 0.05926491692662239, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.6333736777305603, "reward_total_composite_std": 0.22816839814186096} {"timestamp_utc": "2026-04-13T10:55:54Z", "mode": "train", "global_step": 1452, "epoch": 0.14585635359116023, "loss": 0.0033, "grad_norm": 14.863313674926758, "learning_rate": 5.603030303030303e-06, "num_tokens": 2560431.0, "completions/mean_length": 17.625, "completions/min_length": 15.0, "completions/max_length": 20.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 17.625, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 20.0, "rewards/meter/mean": 0.9673625230789185, "rewards/meter/std": 0.04691013693809509, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.7950000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.8443201780319214, "rewards/total_composite/std": 0.1482287347316742, "reward": 0.8443201780319214, "reward_std": 0.1482287347316742, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1140875294804573, "sampling/sampling_logp_difference/max": 1.420192003250122, "sampling/importance_sampling_ratio/min": 0.24166762828826904, "sampling/importance_sampling_ratio/mean": 1.0103696584701538, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6248944774270058, "clip_ratio/low_mean": 0.04411764815449715, "clip_ratio/low_min": 0.04411764815449715, "clip_ratio/high_mean": 0.08975694654509425, "clip_ratio/high_max": 0.08975694654509425, "clip_ratio/region_mean": 0.1338745946995914, "reward_total_mean": 0.8443201780319214, "reward_meter_mean": 0.9673625230789185, "reward_meter_std": 0.04691013693809509, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.7950000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.8443201780319214, "reward_total_composite_std": 0.1482287347316742} {"timestamp_utc": "2026-04-13T10:56:00Z", "mode": "train", "global_step": 1453, "epoch": 0.1459568056253139, "loss": 0.0488, "grad_norm": 11.638545989990234, "learning_rate": 5.600000000000001e-06, "num_tokens": 2562004.0, "completions/mean_length": 40.625, "completions/min_length": 35.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.625, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.6810108423233032, "rewards/meter/std": 0.2987150549888611, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9620885848999023, "rewards/repeat_soft/std": 0.04251923784613609, "rewards/judge_quality/mean": 0.53125, "rewards/judge_quality/std": 0.19334925711154938, "rewards/total_composite/mean": 0.5764018893241882, "rewards/total_composite/std": 0.11793755739927292, "reward": 0.5764018893241882, "reward_std": 0.11793754994869232, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11488058418035507, "sampling/sampling_logp_difference/max": 1.1022356748580933, "sampling/importance_sampling_ratio/min": 0.3321276903152466, "sampling/importance_sampling_ratio/mean": 1.0061787366867065, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6907035112380981, "clip_ratio/low_mean": 0.030543780885636806, "clip_ratio/low_min": 0.030543780885636806, "clip_ratio/high_mean": 0.07158117881044745, "clip_ratio/high_max": 0.07158117881044745, "clip_ratio/region_mean": 0.10212495969608426, "reward_total_mean": 0.5764018893241882, "reward_meter_mean": 0.6810108423233032, "reward_meter_std": 0.2987150549888611, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9620885848999023, "reward_repeat_soft_std": 0.04251923784613609, "reward_judge_quality_mean": 0.53125, "reward_judge_quality_std": 0.19334925711154938, "reward_total_composite_mean": 0.5764018893241882, "reward_total_composite_std": 0.11793755739927292} {"timestamp_utc": "2026-04-13T10:56:07Z", "mode": "train", "global_step": 1454, "epoch": 0.14605725765946762, "loss": -0.0124, "grad_norm": 7.159426689147949, "learning_rate": 5.596969696969697e-06, "num_tokens": 2564393.0, "completions/mean_length": 115.625, "completions/min_length": 109.0, "completions/max_length": 124.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.625, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.9090392589569092, "rewards/meter/std": 0.18287572264671326, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9738054275512695, "rewards/repeat_soft/std": 0.025494271889328957, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.626960039138794, "rewards/total_composite/std": 0.055665262043476105, "reward": 0.626960039138794, "reward_std": 0.05566524714231491, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12418323010206223, "sampling/sampling_logp_difference/max": 2.0376601219177246, "sampling/importance_sampling_ratio/min": 0.1303333193063736, "sampling/importance_sampling_ratio/mean": 1.0259209871292114, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.839865043759346, "clip_ratio/low_mean": 0.11573239602148533, "clip_ratio/low_min": 0.11573239602148533, "clip_ratio/high_mean": 0.012096773833036423, "clip_ratio/high_max": 0.012096773833036423, "clip_ratio/region_mean": 0.12782916985452175, "reward_total_mean": 0.626960039138794, "reward_meter_mean": 0.9090392589569092, "reward_meter_std": 0.18287572264671326, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9738054275512695, "reward_repeat_soft_std": 0.025494271889328957, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.626960039138794, "reward_total_composite_std": 0.055665262043476105} {"timestamp_utc": "2026-04-13T10:56:14Z", "mode": "train", "global_step": 1455, "epoch": 0.1461577096936213, "loss": 0.0213, "grad_norm": 10.604643821716309, "learning_rate": 5.593939393939395e-06, "num_tokens": 2566076.0, "completions/mean_length": 47.375, "completions/min_length": 40.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.375, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.4165652394294739, "rewards/meter/std": 0.3665863275527954, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9656630754470825, "rewards/repeat_soft/std": 0.025733020156621933, "rewards/judge_quality/mean": 0.5649999976158142, "rewards/judge_quality/std": 0.20057062804698944, "rewards/total_composite/mean": 0.5180891752243042, "rewards/total_composite/std": 0.19815339148044586, "reward": 0.5180891752243042, "reward_std": 0.19815339148044586, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1462828516960144, "sampling/sampling_logp_difference/max": 1.6169003248214722, "sampling/importance_sampling_ratio/min": 0.19851306080818176, "sampling/importance_sampling_ratio/mean": 1.005271077156067, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9341767653822899, "clip_ratio/low_mean": 0.07179701700806618, "clip_ratio/low_min": 0.07179701700806618, "clip_ratio/high_mean": 0.07825229875743389, "clip_ratio/high_max": 0.07825229875743389, "clip_ratio/region_mean": 0.15004931576550007, "reward_total_mean": 0.5180891752243042, "reward_meter_mean": 0.4165652394294739, "reward_meter_std": 0.3665863275527954, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9656630754470825, "reward_repeat_soft_std": 0.025733020156621933, "reward_judge_quality_mean": 0.5649999976158142, "reward_judge_quality_std": 0.20057062804698944, "reward_total_composite_mean": 0.5180891752243042, "reward_total_composite_std": 0.19815339148044586} {"timestamp_utc": "2026-04-13T10:56:21Z", "mode": "train", "global_step": 1456, "epoch": 0.14625816172777498, "loss": 0.0332, "grad_norm": 11.490345001220703, "learning_rate": 5.5909090909090915e-06, "num_tokens": 2567633.0, "completions/mean_length": 36.625, "completions/min_length": 32.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.625, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.7870504856109619, "rewards/meter/std": 0.2520248293876648, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9496994614601135, "rewards/repeat_soft/std": 0.06635992974042892, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.5600391626358032, "rewards/total_composite/std": 0.06313219666481018, "reward": 0.5600391626358032, "reward_std": 0.06313218176364899, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08639819175004959, "sampling/sampling_logp_difference/max": 1.2901191711425781, "sampling/importance_sampling_ratio/min": 0.2752379775047302, "sampling/importance_sampling_ratio/mean": 0.9957996010780334, "sampling/importance_sampling_ratio/max": 1.5556946992874146, "entropy": 0.4823068752884865, "clip_ratio/low_mean": 0.044285462237894535, "clip_ratio/low_min": 0.044285462237894535, "clip_ratio/high_mean": 0.05146470386534929, "clip_ratio/high_max": 0.05146470386534929, "clip_ratio/region_mean": 0.09575016610324383, "reward_total_mean": 0.5600391626358032, "reward_meter_mean": 0.7870504856109619, "reward_meter_std": 0.2520248293876648, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9496994614601135, "reward_repeat_soft_std": 0.06635992974042892, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.5600391626358032, "reward_total_composite_std": 0.06313219666481018} {"timestamp_utc": "2026-04-13T10:56:28Z", "mode": "train", "global_step": 1457, "epoch": 0.14635861376192869, "loss": 0.042, "grad_norm": 11.235291481018066, "learning_rate": 5.587878787878789e-06, "num_tokens": 2569854.0, "completions/mean_length": 98.625, "completions/min_length": 87.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.625, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.7525693774223328, "rewards/meter/std": 0.2684197425842285, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9529146552085876, "rewards/repeat_soft/std": 0.03146170824766159, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.5393280982971191, "rewards/total_composite/std": 0.0792858675122261, "reward": 0.5393280982971191, "reward_std": 0.07928586006164551, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1270921230316162, "sampling/sampling_logp_difference/max": 1.6169508695602417, "sampling/importance_sampling_ratio/min": 0.1985030472278595, "sampling/importance_sampling_ratio/mean": 1.026190996170044, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7462729886174202, "clip_ratio/low_mean": 0.043929059989750385, "clip_ratio/low_min": 0.043929059989750385, "clip_ratio/high_mean": 0.05161045212298632, "clip_ratio/high_max": 0.05161045212298632, "clip_ratio/region_mean": 0.0955395121127367, "reward_total_mean": 0.5393280982971191, "reward_meter_mean": 0.7525693774223328, "reward_meter_std": 0.2684197425842285, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9529146552085876, "reward_repeat_soft_std": 0.03146170824766159, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.5393280982971191, "reward_total_composite_std": 0.0792858675122261} {"timestamp_utc": "2026-04-13T10:56:34Z", "mode": "train", "global_step": 1458, "epoch": 0.14645906579608237, "loss": 0.043, "grad_norm": 7.706801891326904, "learning_rate": 5.584848484848485e-06, "num_tokens": 2572174.0, "completions/mean_length": 88.0, "completions/min_length": 80.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 88.0, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.7622480392456055, "rewards/meter/std": 0.2639716863632202, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8169191479682922, "rewards/repeat_soft/std": 0.05823111906647682, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5306315422058105, "rewards/total_composite/std": 0.06933984160423279, "reward": 0.5306315422058105, "reward_std": 0.06933984160423279, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08421962708234787, "sampling/sampling_logp_difference/max": 1.8180437088012695, "sampling/importance_sampling_ratio/min": 0.16234302520751953, "sampling/importance_sampling_ratio/mean": 1.0130640268325806, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.43276629596948624, "clip_ratio/low_mean": 0.030778031796216965, "clip_ratio/low_min": 0.030778031796216965, "clip_ratio/high_mean": 0.05446739564649761, "clip_ratio/high_max": 0.05446739564649761, "clip_ratio/region_mean": 0.08524542744271457, "reward_total_mean": 0.5306315422058105, "reward_meter_mean": 0.7622480392456055, "reward_meter_std": 0.2639716863632202, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8169191479682922, "reward_repeat_soft_std": 0.05823111906647682, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5306315422058105, "reward_total_composite_std": 0.06933984160423279} {"timestamp_utc": "2026-04-13T10:56:40Z", "mode": "train", "global_step": 1459, "epoch": 0.14655951783023607, "loss": 0.0239, "grad_norm": 12.355224609375, "learning_rate": 5.5818181818181824e-06, "num_tokens": 2573928.0, "completions/mean_length": 44.25, "completions/min_length": 39.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.25, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.6194192171096802, "rewards/meter/std": 0.3770742416381836, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9371640682220459, "rewards/repeat_soft/std": 0.05107033997774124, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.5168610215187073, "rewards/total_composite/std": 0.1104142963886261, "reward": 0.5168610215187073, "reward_std": 0.1104142889380455, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10699944943189621, "sampling/sampling_logp_difference/max": 1.3431096076965332, "sampling/importance_sampling_ratio/min": 0.26103270053863525, "sampling/importance_sampling_ratio/mean": 1.0050694942474365, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5965271331369877, "clip_ratio/low_mean": 0.03899620100855827, "clip_ratio/low_min": 0.03899620100855827, "clip_ratio/high_mean": 0.056769791059195995, "clip_ratio/high_max": 0.056769791059195995, "clip_ratio/region_mean": 0.09576599206775427, "reward_total_mean": 0.5168610215187073, "reward_meter_mean": 0.6194192171096802, "reward_meter_std": 0.3770742416381836, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9371640682220459, "reward_repeat_soft_std": 0.05107033997774124, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.5168610215187073, "reward_total_composite_std": 0.1104142963886261} {"timestamp_utc": "2026-04-13T10:56:47Z", "mode": "train", "global_step": 1460, "epoch": 0.14665996986438976, "loss": -0.0616, "grad_norm": 11.842880249023438, "learning_rate": 5.578787878787879e-06, "num_tokens": 2576086.0, "completions/mean_length": 92.75, "completions/min_length": 66.0, "completions/max_length": 124.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.75, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.05187898874282837, "rewards/meter/std": 0.07328092306852341, "rewards/count_adherence/mean": 0.8250000476837158, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.966722846031189, "rewards/repeat_soft/std": 0.021765653043985367, "rewards/judge_quality/mean": 0.5187499523162842, "rewards/judge_quality/std": 0.17908000946044922, "rewards/total_composite/mean": 0.33348724246025085, "rewards/total_composite/std": 0.04362621530890465, "reward": 0.33348724246025085, "reward_std": 0.04362621530890465, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15613049268722534, "sampling/sampling_logp_difference/max": 3.2224905490875244, "sampling/importance_sampling_ratio/min": 0.03985567018389702, "sampling/importance_sampling_ratio/mean": 0.9895204305648804, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5575118139386177, "clip_ratio/low_mean": 0.10854641254991293, "clip_ratio/low_min": 0.10854641254991293, "clip_ratio/high_mean": 0.03530109953135252, "clip_ratio/high_max": 0.03530109953135252, "clip_ratio/region_mean": 0.14384751208126545, "reward_total_mean": 0.33348724246025085, "reward_meter_mean": 0.05187898874282837, "reward_meter_std": 0.07328092306852341, "reward_count_adherence_mean": 0.8250000476837158, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.966722846031189, "reward_repeat_soft_std": 0.021765653043985367, "reward_judge_quality_mean": 0.5187499523162842, "reward_judge_quality_std": 0.17908000946044922, "reward_total_composite_mean": 0.33348724246025085, "reward_total_composite_std": 0.04362621530890465} {"timestamp_utc": "2026-04-13T10:56:54Z", "mode": "train", "global_step": 1461, "epoch": 0.14676042189854344, "loss": 0.0691, "grad_norm": 20.209810256958008, "learning_rate": 5.575757575757577e-06, "num_tokens": 2577949.0, "completions/mean_length": 53.875, "completions/min_length": 43.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.875, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9132522940635681, "rewards/meter/std": 0.17970865964889526, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8632922172546387, "rewards/repeat_soft/std": 0.07984157651662827, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.5811365842819214, "rewards/total_composite/std": 0.04944567382335663, "reward": 0.5811365842819214, "reward_std": 0.04944567382335663, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11274424195289612, "sampling/sampling_logp_difference/max": 1.6341705322265625, "sampling/importance_sampling_ratio/min": 0.1951141357421875, "sampling/importance_sampling_ratio/mean": 1.0093806982040405, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6105680763721466, "clip_ratio/low_mean": 0.01848655566573143, "clip_ratio/low_min": 0.01848655566573143, "clip_ratio/high_mean": 0.06229339726269245, "clip_ratio/high_max": 0.06229339726269245, "clip_ratio/region_mean": 0.08077995292842388, "reward_total_mean": 0.5811365842819214, "reward_meter_mean": 0.9132522940635681, "reward_meter_std": 0.17970865964889526, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8632922172546387, "reward_repeat_soft_std": 0.07984157651662827, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.5811365842819214, "reward_total_composite_std": 0.04944567382335663} {"timestamp_utc": "2026-04-13T10:57:00Z", "mode": "train", "global_step": 1462, "epoch": 0.14686087393269714, "loss": 0.0297, "grad_norm": 11.300406455993652, "learning_rate": 5.572727272727273e-06, "num_tokens": 2579675.0, "completions/mean_length": 59.75, "completions/min_length": 51.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.75, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.4085647165775299, "rewards/meter/std": 0.404621422290802, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.907824695110321, "rewards/repeat_soft/std": 0.06312815845012665, "rewards/judge_quality/mean": 0.45624998211860657, "rewards/judge_quality/std": 0.18126046657562256, "rewards/total_composite/mean": 0.46732097864151, "rewards/total_composite/std": 0.16994957625865936, "reward": 0.46732097864151, "reward_std": 0.16994957625865936, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13315221667289734, "sampling/sampling_logp_difference/max": 3.153635263442993, "sampling/importance_sampling_ratio/min": 0.04269663244485855, "sampling/importance_sampling_ratio/mean": 1.0035516023635864, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6467822268605232, "clip_ratio/low_mean": 0.08823794219642878, "clip_ratio/low_min": 0.08823794219642878, "clip_ratio/high_mean": 0.03578192740678787, "clip_ratio/high_max": 0.03578192740678787, "clip_ratio/region_mean": 0.12401986960321665, "reward_total_mean": 0.46732097864151, "reward_meter_mean": 0.4085647165775299, "reward_meter_std": 0.404621422290802, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.907824695110321, "reward_repeat_soft_std": 0.06312815845012665, "reward_judge_quality_mean": 0.45624998211860657, "reward_judge_quality_std": 0.18126046657562256, "reward_total_composite_mean": 0.46732097864151, "reward_total_composite_std": 0.16994957625865936} {"timestamp_utc": "2026-04-13T10:57:06Z", "mode": "train", "global_step": 1463, "epoch": 0.14696132596685083, "loss": 0.0272, "grad_norm": 13.791470527648926, "learning_rate": 5.569696969696971e-06, "num_tokens": 2581272.0, "completions/mean_length": 29.625, "completions/min_length": 24.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.625, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.7594460248947144, "rewards/meter/std": 0.2380126267671585, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8953207731246948, "rewards/repeat_soft/std": 0.11132125556468964, "rewards/judge_quality/mean": 0.3675000071525574, "rewards/judge_quality/std": 0.1348809003829956, "rewards/total_composite/mean": 0.5211995840072632, "rewards/total_composite/std": 0.10588626563549042, "reward": 0.5211995840072632, "reward_std": 0.10588628053665161, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12958385050296783, "sampling/sampling_logp_difference/max": 1.413332223892212, "sampling/importance_sampling_ratio/min": 0.24333108961582184, "sampling/importance_sampling_ratio/mean": 0.9924678802490234, "sampling/importance_sampling_ratio/max": 1.768438458442688, "entropy": 0.7009607888758183, "clip_ratio/low_mean": 0.058047713711857796, "clip_ratio/low_min": 0.058047713711857796, "clip_ratio/high_mean": 0.06673027109354734, "clip_ratio/high_max": 0.06673027109354734, "clip_ratio/region_mean": 0.12477798480540514, "reward_total_mean": 0.5211995840072632, "reward_meter_mean": 0.7594460248947144, "reward_meter_std": 0.2380126267671585, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8953207731246948, "reward_repeat_soft_std": 0.11132125556468964, "reward_judge_quality_mean": 0.3675000071525574, "reward_judge_quality_std": 0.1348809003829956, "reward_total_composite_mean": 0.5211995840072632, "reward_total_composite_std": 0.10588626563549042} {"timestamp_utc": "2026-04-13T10:57:13Z", "mode": "train", "global_step": 1464, "epoch": 0.14706177800100453, "loss": 0.044, "grad_norm": 9.621746063232422, "learning_rate": 5.566666666666667e-06, "num_tokens": 2582894.0, "completions/mean_length": 38.75, "completions/min_length": 34.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.75, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.9511234164237976, "rewards/meter/std": 0.10077424347400665, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9086436033248901, "rewards/repeat_soft/std": 0.08281069248914719, "rewards/judge_quality/mean": 0.8575000166893005, "rewards/judge_quality/std": 0.176776722073555, "rewards/total_composite/mean": 0.8650649785995483, "rewards/total_composite/std": 0.12106508016586304, "reward": 0.8650649785995483, "reward_std": 0.12106508016586304, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09151957184076309, "sampling/sampling_logp_difference/max": 1.1939563751220703, "sampling/importance_sampling_ratio/min": 0.30302003026008606, "sampling/importance_sampling_ratio/mean": 1.0106608867645264, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5162233076989651, "clip_ratio/low_mean": 0.02119965897873044, "clip_ratio/low_min": 0.02119965897873044, "clip_ratio/high_mean": 0.09480717871338129, "clip_ratio/high_max": 0.09480717871338129, "clip_ratio/region_mean": 0.11600683769211173, "reward_total_mean": 0.8650649785995483, "reward_meter_mean": 0.9511234164237976, "reward_meter_std": 0.10077424347400665, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9086436033248901, "reward_repeat_soft_std": 0.08281069248914719, "reward_judge_quality_mean": 0.8575000166893005, "reward_judge_quality_std": 0.176776722073555, "reward_total_composite_mean": 0.8650649785995483, "reward_total_composite_std": 0.12106508016586304} {"timestamp_utc": "2026-04-13T10:57:19Z", "mode": "train", "global_step": 1465, "epoch": 0.14716223003515821, "loss": -0.0256, "grad_norm": 12.525416374206543, "learning_rate": 5.563636363636364e-06, "num_tokens": 2584391.0, "completions/mean_length": 42.125, "completions/min_length": 35.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.125, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9827646613121033, "rewards/meter/std": 0.017621733248233795, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9358309507369995, "rewards/repeat_soft/std": 0.02582893893122673, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.6886136531829834, "rewards/total_composite/std": 0.14731666445732117, "reward": 0.6886136531829834, "reward_std": 0.14731664955615997, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.105953149497509, "sampling/sampling_logp_difference/max": 1.130956768989563, "sampling/importance_sampling_ratio/min": 0.3227243423461914, "sampling/importance_sampling_ratio/mean": 1.01067054271698, "sampling/importance_sampling_ratio/max": 1.8950624465942383, "entropy": 0.568340964615345, "clip_ratio/low_mean": 0.07531766872853041, "clip_ratio/low_min": 0.07531766872853041, "clip_ratio/high_mean": 0.027173913083970547, "clip_ratio/high_max": 0.027173913083970547, "clip_ratio/region_mean": 0.10249158181250095, "reward_total_mean": 0.6886136531829834, "reward_meter_mean": 0.9827646613121033, "reward_meter_std": 0.017621733248233795, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9358309507369995, "reward_repeat_soft_std": 0.02582893893122673, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.6886136531829834, "reward_total_composite_std": 0.14731666445732117} {"timestamp_utc": "2026-04-13T10:57:26Z", "mode": "train", "global_step": 1466, "epoch": 0.1472626820693119, "loss": 0.0033, "grad_norm": 11.49193000793457, "learning_rate": 5.560606060606061e-06, "num_tokens": 2585914.0, "completions/mean_length": 47.375, "completions/min_length": 40.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.375, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.664898157119751, "rewards/meter/std": 0.3572017252445221, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9430370330810547, "rewards/repeat_soft/std": 0.09419349581003189, "rewards/judge_quality/mean": 0.53125, "rewards/judge_quality/std": 0.1865811049938202, "rewards/total_composite/mean": 0.537501335144043, "rewards/total_composite/std": 0.07693111896514893, "reward": 0.537501335144043, "reward_std": 0.07693111896514893, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12955878674983978, "sampling/sampling_logp_difference/max": 3.5795273780822754, "sampling/importance_sampling_ratio/min": 0.02788887731730938, "sampling/importance_sampling_ratio/mean": 1.0106037855148315, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7453957945108414, "clip_ratio/low_mean": 0.0381249999627471, "clip_ratio/low_min": 0.0381249999627471, "clip_ratio/high_mean": 0.07593381218612194, "clip_ratio/high_max": 0.07593381218612194, "clip_ratio/region_mean": 0.11405881214886904, "reward_total_mean": 0.537501335144043, "reward_meter_mean": 0.664898157119751, "reward_meter_std": 0.3572017252445221, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9430370330810547, "reward_repeat_soft_std": 0.09419349581003189, "reward_judge_quality_mean": 0.53125, "reward_judge_quality_std": 0.1865811049938202, "reward_total_composite_mean": 0.537501335144043, "reward_total_composite_std": 0.07693111896514893} {"timestamp_utc": "2026-04-13T10:57:33Z", "mode": "train", "global_step": 1467, "epoch": 0.1473631341034656, "loss": -0.0289, "grad_norm": 8.926210403442383, "learning_rate": 5.557575757575758e-06, "num_tokens": 2587774.0, "completions/mean_length": 60.5, "completions/min_length": 47.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.5, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.7280633449554443, "rewards/meter/std": 0.30291086435317993, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8180413842201233, "rewards/repeat_soft/std": 0.08221400529146194, "rewards/judge_quality/mean": 0.5237500071525574, "rewards/judge_quality/std": 0.25150617957115173, "rewards/total_composite/mean": 0.5870155692100525, "rewards/total_composite/std": 0.21065010130405426, "reward": 0.5870155692100525, "reward_std": 0.21065007150173187, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11216025054454803, "sampling/sampling_logp_difference/max": 1.7116155624389648, "sampling/importance_sampling_ratio/min": 0.18057382106781006, "sampling/importance_sampling_ratio/mean": 1.0145072937011719, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6791957169771194, "clip_ratio/low_mean": 0.07617599982768297, "clip_ratio/low_min": 0.07617599982768297, "clip_ratio/high_mean": 0.02996880654245615, "clip_ratio/high_max": 0.02996880654245615, "clip_ratio/region_mean": 0.10614480637013912, "reward_total_mean": 0.5870155692100525, "reward_meter_mean": 0.7280633449554443, "reward_meter_std": 0.30291086435317993, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8180413842201233, "reward_repeat_soft_std": 0.08221400529146194, "reward_judge_quality_mean": 0.5237500071525574, "reward_judge_quality_std": 0.25150617957115173, "reward_total_composite_mean": 0.5870155692100525, "reward_total_composite_std": 0.21065010130405426} {"timestamp_utc": "2026-04-13T10:57:39Z", "mode": "train", "global_step": 1468, "epoch": 0.14746358613761928, "loss": -0.0245, "grad_norm": 16.370771408081055, "learning_rate": 5.554545454545454e-06, "num_tokens": 2589059.0, "completions/mean_length": 23.625, "completions/min_length": 20.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.625, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9890918731689453, "rewards/meter/std": 0.00959152914583683, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9175378680229187, "rewards/repeat_soft/std": 0.11084415018558502, "rewards/judge_quality/mean": 0.3462499976158142, "rewards/judge_quality/std": 0.10336308926343918, "rewards/total_composite/mean": 0.5601323843002319, "rewards/total_composite/std": 0.061217617243528366, "reward": 0.5601323843002319, "reward_std": 0.06121760234236717, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1455719918012619, "sampling/sampling_logp_difference/max": 2.6551313400268555, "sampling/importance_sampling_ratio/min": 0.07028961181640625, "sampling/importance_sampling_ratio/mean": 0.9993245601654053, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.094538040459156, "clip_ratio/low_mean": 0.06639610603451729, "clip_ratio/low_min": 0.06639610603451729, "clip_ratio/high_mean": 0.06223659124225378, "clip_ratio/high_max": 0.06223659124225378, "clip_ratio/region_mean": 0.12863269727677107, "reward_total_mean": 0.5601323843002319, "reward_meter_mean": 0.9890918731689453, "reward_meter_std": 0.00959152914583683, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9175378680229187, "reward_repeat_soft_std": 0.11084415018558502, "reward_judge_quality_mean": 0.3462499976158142, "reward_judge_quality_std": 0.10336308926343918, "reward_total_composite_mean": 0.5601323843002319, "reward_total_composite_std": 0.061217617243528366} {"timestamp_utc": "2026-04-13T10:57:47Z", "mode": "train", "global_step": 1469, "epoch": 0.14756403817177297, "loss": 0.0148, "grad_norm": 6.228905200958252, "learning_rate": 5.5515151515151524e-06, "num_tokens": 2591513.0, "completions/mean_length": 129.75, "completions/min_length": 100.0, "completions/max_length": 154.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 129.75, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 154.0, "rewards/meter/mean": 0.7403182983398438, "rewards/meter/std": 0.1365850269794464, "rewards/count_adherence/mean": 0.8541666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7926737070083618, "rewards/repeat_soft/std": 0.11876393854618073, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.10966669768095016, "rewards/total_composite/mean": 0.4557223618030548, "rewards/total_composite/std": 0.06416784226894379, "reward": 0.4557223618030548, "reward_std": 0.06416784226894379, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11837196350097656, "sampling/sampling_logp_difference/max": 1.9244991540908813, "sampling/importance_sampling_ratio/min": 0.1459488421678543, "sampling/importance_sampling_ratio/mean": 1.000449538230896, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6611907668411732, "clip_ratio/low_mean": 0.06312037538737059, "clip_ratio/low_min": 0.06312037538737059, "clip_ratio/high_mean": 0.05303825717419386, "clip_ratio/high_max": 0.05303825717419386, "clip_ratio/region_mean": 0.11615863256156445, "reward_total_mean": 0.4557223618030548, "reward_meter_mean": 0.7403182983398438, "reward_meter_std": 0.1365850269794464, "reward_count_adherence_mean": 0.8541666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7926737070083618, "reward_repeat_soft_std": 0.11876393854618073, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.10966669768095016, "reward_total_composite_mean": 0.4557223618030548, "reward_total_composite_std": 0.06416784226894379} {"timestamp_utc": "2026-04-13T10:57:54Z", "mode": "train", "global_step": 1470, "epoch": 0.14766449020592667, "loss": 0.0308, "grad_norm": 7.634474277496338, "learning_rate": 5.548484848484849e-06, "num_tokens": 2594005.0, "completions/mean_length": 116.5, "completions/min_length": 90.0, "completions/max_length": 137.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 116.5, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 137.0, "rewards/meter/mean": 0.7753080129623413, "rewards/meter/std": 0.27682748436927795, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9754493236541748, "rewards/repeat_soft/std": 0.017098110169172287, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.4989067316055298, "rewards/total_composite/std": 0.08452694118022919, "reward": 0.4989067316055298, "reward_std": 0.08452694863080978, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14233583211898804, "sampling/sampling_logp_difference/max": 1.5213804244995117, "sampling/importance_sampling_ratio/min": 0.2184101939201355, "sampling/importance_sampling_ratio/mean": 1.0129231214523315, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0378146320581436, "clip_ratio/low_mean": 0.06215149536728859, "clip_ratio/low_min": 0.06215149536728859, "clip_ratio/high_mean": 0.07031365670263767, "clip_ratio/high_max": 0.07031365670263767, "clip_ratio/region_mean": 0.13246515206992626, "reward_total_mean": 0.4989067316055298, "reward_meter_mean": 0.7753080129623413, "reward_meter_std": 0.27682748436927795, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9754493236541748, "reward_repeat_soft_std": 0.017098110169172287, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.4989067316055298, "reward_total_composite_std": 0.08452694118022919} {"timestamp_utc": "2026-04-13T10:58:00Z", "mode": "train", "global_step": 1471, "epoch": 0.14776494224008035, "loss": 0.0236, "grad_norm": 13.744756698608398, "learning_rate": 5.545454545454546e-06, "num_tokens": 2595614.0, "completions/mean_length": 48.125, "completions/min_length": 43.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.125, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.6877424716949463, "rewards/meter/std": 0.4287676513195038, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.892635703086853, "rewards/repeat_soft/std": 0.08857487142086029, "rewards/judge_quality/mean": 0.6987500190734863, "rewards/judge_quality/std": 0.31651848554611206, "rewards/total_composite/mean": 0.6591956615447998, "rewards/total_composite/std": 0.28758805990219116, "reward": 0.6591956615447998, "reward_std": 0.28758805990219116, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13303479552268982, "sampling/sampling_logp_difference/max": 1.1878950595855713, "sampling/importance_sampling_ratio/min": 0.30486229062080383, "sampling/importance_sampling_ratio/mean": 1.0081233978271484, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8316401988267899, "clip_ratio/low_mean": 0.05475694499909878, "clip_ratio/low_min": 0.05475694499909878, "clip_ratio/high_mean": 0.06690456345677376, "clip_ratio/high_max": 0.06690456345677376, "clip_ratio/region_mean": 0.12166150845587254, "reward_total_mean": 0.6591956615447998, "reward_meter_mean": 0.6877424716949463, "reward_meter_std": 0.4287676513195038, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.892635703086853, "reward_repeat_soft_std": 0.08857487142086029, "reward_judge_quality_mean": 0.6987500190734863, "reward_judge_quality_std": 0.31651848554611206, "reward_total_composite_mean": 0.6591956615447998, "reward_total_composite_std": 0.28758805990219116} {"timestamp_utc": "2026-04-13T10:58:06Z", "mode": "train", "global_step": 1472, "epoch": 0.14786539427423406, "loss": -0.0372, "grad_norm": 12.051742553710938, "learning_rate": 5.5424242424242425e-06, "num_tokens": 2597271.0, "completions/mean_length": 43.125, "completions/min_length": 34.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.125, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.8499079346656799, "rewards/meter/std": 0.3276419937610626, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8506057858467102, "rewards/repeat_soft/std": 0.047927580773830414, "rewards/judge_quality/mean": 0.4137499928474426, "rewards/judge_quality/std": 0.06781013309955597, "rewards/total_composite/mean": 0.5555394887924194, "rewards/total_composite/std": 0.09663747251033783, "reward": 0.5555394887924194, "reward_std": 0.09663745760917664, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1002088412642479, "sampling/sampling_logp_difference/max": 1.6085458993911743, "sampling/importance_sampling_ratio/min": 0.20017847418785095, "sampling/importance_sampling_ratio/mean": 1.0045076608657837, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.625467523932457, "clip_ratio/low_mean": 0.01907442230731249, "clip_ratio/low_min": 0.01907442230731249, "clip_ratio/high_mean": 0.06602906063199043, "clip_ratio/high_max": 0.06602906063199043, "clip_ratio/region_mean": 0.08510348293930292, "reward_total_mean": 0.5555394887924194, "reward_meter_mean": 0.8499079346656799, "reward_meter_std": 0.3276419937610626, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8506057858467102, "reward_repeat_soft_std": 0.047927580773830414, "reward_judge_quality_mean": 0.4137499928474426, "reward_judge_quality_std": 0.06781013309955597, "reward_total_composite_mean": 0.5555394887924194, "reward_total_composite_std": 0.09663747251033783} {"timestamp_utc": "2026-04-13T10:58:12Z", "mode": "train", "global_step": 1473, "epoch": 0.14796584630838774, "loss": 0.078, "grad_norm": 17.361730575561523, "learning_rate": 5.53939393939394e-06, "num_tokens": 2598815.0, "completions/mean_length": 35.0, "completions/min_length": 31.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.0, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.3028879761695862, "rewards/meter/std": 0.32608795166015625, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9703353643417358, "rewards/repeat_soft/std": 0.058450888842344284, "rewards/judge_quality/mean": 0.7950000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.5166357755661011, "rewards/total_composite/std": 0.20340381562709808, "reward": 0.5166357755661011, "reward_std": 0.2034038007259369, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11877524852752686, "sampling/sampling_logp_difference/max": 1.2922561168670654, "sampling/importance_sampling_ratio/min": 0.2746504247188568, "sampling/importance_sampling_ratio/mean": 1.0062525272369385, "sampling/importance_sampling_ratio/max": 1.9707258939743042, "entropy": 0.5405756812542677, "clip_ratio/low_mean": 0.0628345743753016, "clip_ratio/low_min": 0.0628345743753016, "clip_ratio/high_mean": 0.038352273404598236, "clip_ratio/high_max": 0.038352273404598236, "clip_ratio/region_mean": 0.10118684777989984, "reward_total_mean": 0.5166357755661011, "reward_meter_mean": 0.3028879761695862, "reward_meter_std": 0.32608795166015625, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9703353643417358, "reward_repeat_soft_std": 0.058450888842344284, "reward_judge_quality_mean": 0.7950000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.5166357755661011, "reward_total_composite_std": 0.20340381562709808} {"timestamp_utc": "2026-04-13T10:58:19Z", "mode": "train", "global_step": 1474, "epoch": 0.14806629834254142, "loss": 0.0647, "grad_norm": 13.33869457244873, "learning_rate": 5.536363636363636e-06, "num_tokens": 2600649.0, "completions/mean_length": 54.25, "completions/min_length": 42.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.25, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.4765779674053192, "rewards/meter/std": 0.3976920247077942, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9742038249969482, "rewards/repeat_soft/std": 0.027118271216750145, "rewards/judge_quality/mean": 0.46875, "rewards/judge_quality/std": 0.19334925711154938, "rewards/total_composite/mean": 0.4688124358654022, "rewards/total_composite/std": 0.26618191599845886, "reward": 0.4688124358654022, "reward_std": 0.2661818861961365, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15565451979637146, "sampling/sampling_logp_difference/max": 1.6274919509887695, "sampling/importance_sampling_ratio/min": 0.1964215785264969, "sampling/importance_sampling_ratio/mean": 1.0438464879989624, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3133893460035324, "clip_ratio/low_mean": 0.09236462041735649, "clip_ratio/low_min": 0.09236462041735649, "clip_ratio/high_mean": 0.045936762355268, "clip_ratio/high_max": 0.045936762355268, "clip_ratio/region_mean": 0.1383013827726245, "reward_total_mean": 0.4688124358654022, "reward_meter_mean": 0.4765779674053192, "reward_meter_std": 0.3976920247077942, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9742038249969482, "reward_repeat_soft_std": 0.027118271216750145, "reward_judge_quality_mean": 0.46875, "reward_judge_quality_std": 0.19334925711154938, "reward_total_composite_mean": 0.4688124358654022, "reward_total_composite_std": 0.26618191599845886} {"timestamp_utc": "2026-04-13T10:58:27Z", "mode": "train", "global_step": 1475, "epoch": 0.14816675037669513, "loss": 0.0295, "grad_norm": 10.713043212890625, "learning_rate": 5.533333333333334e-06, "num_tokens": 2603073.0, "completions/mean_length": 104.0, "completions/min_length": 81.0, "completions/max_length": 125.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 104.0, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.648520827293396, "rewards/meter/std": 0.2619701325893402, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.1035098284482956, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9017126560211182, "rewards/repeat_soft/std": 0.07726913690567017, "rewards/judge_quality/mean": 0.5575000047683716, "rewards/judge_quality/std": 0.19955308735370636, "rewards/total_composite/mean": 0.5715631246566772, "rewards/total_composite/std": 0.18075372278690338, "reward": 0.5715631246566772, "reward_std": 0.18075372278690338, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09615106880664825, "sampling/sampling_logp_difference/max": 3.520033359527588, "sampling/importance_sampling_ratio/min": 0.029598446562886238, "sampling/importance_sampling_ratio/mean": 1.006070852279663, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3830401562154293, "clip_ratio/low_mean": 0.06448802212253213, "clip_ratio/low_min": 0.06448802212253213, "clip_ratio/high_mean": 0.01771604921668768, "clip_ratio/high_max": 0.01771604921668768, "clip_ratio/region_mean": 0.08220407133921981, "reward_total_mean": 0.5715631246566772, "reward_meter_mean": 0.648520827293396, "reward_meter_std": 0.2619701325893402, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.1035098284482956, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9017126560211182, "reward_repeat_soft_std": 0.07726913690567017, "reward_judge_quality_mean": 0.5575000047683716, "reward_judge_quality_std": 0.19955308735370636, "reward_total_composite_mean": 0.5715631246566772, "reward_total_composite_std": 0.18075372278690338} {"timestamp_utc": "2026-04-13T10:58:38Z", "mode": "train", "global_step": 1476, "epoch": 0.14826720241084881, "loss": -0.1071, "grad_norm": 2.849167585372925, "learning_rate": 5.530303030303031e-06, "num_tokens": 2604754.0, "completions/mean_length": 109.125, "completions/min_length": 40.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 51.57143020629883, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.7600843906402588, "rewards/meter/std": 0.2595258355140686, "rewards/count_adherence/mean": 0.8333333730697632, "rewards/count_adherence/std": 0.2519763112068176, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8423864841461182, "rewards/repeat_soft/std": 0.12611058354377747, "rewards/judge_quality/mean": 0.3399999737739563, "rewards/judge_quality/std": 0.15052290260791779, "rewards/total_composite/mean": 0.44253116846084595, "rewards/total_composite/std": 0.20070195198059082, "reward": 0.44253116846084595, "reward_std": 0.20070193707942963, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10046947747468948, "sampling/sampling_logp_difference/max": 1.9943890571594238, "sampling/importance_sampling_ratio/min": 0.1360967755317688, "sampling/importance_sampling_ratio/mean": 0.9980024099349976, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3889933116734028, "clip_ratio/low_mean": 0.005859375, "clip_ratio/low_min": 0.005859375, "clip_ratio/high_mean": 0.07746341777965426, "clip_ratio/high_max": 0.07746341777965426, "clip_ratio/region_mean": 0.08332279277965426, "reward_total_mean": 0.44253116846084595, "reward_meter_mean": 0.7600843906402588, "reward_meter_std": 0.2595258355140686, "reward_count_adherence_mean": 0.8333333730697632, "reward_count_adherence_std": 0.2519763112068176, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8423864841461182, "reward_repeat_soft_std": 0.12611058354377747, "reward_judge_quality_mean": 0.3399999737739563, "reward_judge_quality_std": 0.15052290260791779, "reward_total_composite_mean": 0.44253116846084595, "reward_total_composite_std": 0.20070195198059082} {"timestamp_utc": "2026-04-13T10:58:46Z", "mode": "train", "global_step": 1477, "epoch": 0.14836765444500252, "loss": -0.0015, "grad_norm": 6.90681791305542, "learning_rate": 5.527272727272728e-06, "num_tokens": 2607584.0, "completions/mean_length": 128.75, "completions/min_length": 101.0, "completions/max_length": 169.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 128.75, "completions/min_terminated_length": 101.0, "completions/max_terminated_length": 169.0, "rewards/meter/mean": 0.33890193700790405, "rewards/meter/std": 0.31740322709083557, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.07715168595314026, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8658617734909058, "rewards/repeat_soft/std": 0.06393294781446457, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.40142741799354553, "rewards/total_composite/std": 0.08093750476837158, "reward": 0.40142741799354553, "reward_std": 0.08093750476837158, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13084861636161804, "sampling/sampling_logp_difference/max": 4.246708869934082, "sampling/importance_sampling_ratio/min": 0.014311256818473339, "sampling/importance_sampling_ratio/mean": 0.9870463013648987, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6351949870586395, "clip_ratio/low_mean": 0.06428987253457308, "clip_ratio/low_min": 0.06428987253457308, "clip_ratio/high_mean": 0.05529332160949707, "clip_ratio/high_max": 0.05529332160949707, "clip_ratio/region_mean": 0.11958319414407015, "reward_total_mean": 0.40142741799354553, "reward_meter_mean": 0.33890193700790405, "reward_meter_std": 0.31740322709083557, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.07715168595314026, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8658617734909058, "reward_repeat_soft_std": 0.06393294781446457, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.40142741799354553, "reward_total_composite_std": 0.08093750476837158} {"timestamp_utc": "2026-04-13T10:58:57Z", "mode": "train", "global_step": 1478, "epoch": 0.1484681064791562, "loss": -0.1887, "grad_norm": 1.7883447408676147, "learning_rate": 5.524242424242424e-06, "num_tokens": 2609757.0, "completions/mean_length": 144.625, "completions/min_length": 83.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 92.14286041259766, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.8850111961364746, "rewards/meter/std": 0.1782539188861847, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8292892575263977, "rewards/repeat_soft/std": 0.0910106971859932, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.4554700553417206, "rewards/total_composite/std": 0.18945030868053436, "reward": 0.4554700553417206, "reward_std": 0.18945029377937317, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10043580830097198, "sampling/sampling_logp_difference/max": 2.3448104858398438, "sampling/importance_sampling_ratio/min": 0.09586536884307861, "sampling/importance_sampling_ratio/mean": 1.0046113729476929, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5433511883020401, "clip_ratio/low_mean": 0.006793478038161993, "clip_ratio/low_min": 0.006793478038161993, "clip_ratio/high_mean": 0.10017213178798556, "clip_ratio/high_max": 0.10017213178798556, "clip_ratio/region_mean": 0.10696560982614756, "reward_total_mean": 0.4554700553417206, "reward_meter_mean": 0.8850111961364746, "reward_meter_std": 0.1782539188861847, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8292892575263977, "reward_repeat_soft_std": 0.0910106971859932, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.4554700553417206, "reward_total_composite_std": 0.18945030868053436} {"timestamp_utc": "2026-04-13T10:59:03Z", "mode": "train", "global_step": 1479, "epoch": 0.14856855851330988, "loss": 0.1153, "grad_norm": 14.996219635009766, "learning_rate": 5.521212121212122e-06, "num_tokens": 2611292.0, "completions/mean_length": 44.875, "completions/min_length": 35.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.875, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.8259141445159912, "rewards/meter/std": 0.26048386096954346, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9079709053039551, "rewards/repeat_soft/std": 0.06270492821931839, "rewards/judge_quality/mean": 0.5237500071525574, "rewards/judge_quality/std": 0.19078317284584045, "rewards/total_composite/mean": 0.6216350197792053, "rewards/total_composite/std": 0.1506887674331665, "reward": 0.6216350197792053, "reward_std": 0.1506887674331665, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12402523308992386, "sampling/sampling_logp_difference/max": 2.003143310546875, "sampling/importance_sampling_ratio/min": 0.13491055369377136, "sampling/importance_sampling_ratio/mean": 0.9949265718460083, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7351835444569588, "clip_ratio/low_mean": 0.06681718723848462, "clip_ratio/low_min": 0.06681718723848462, "clip_ratio/high_mean": 0.031944445334374905, "clip_ratio/high_max": 0.031944445334374905, "clip_ratio/region_mean": 0.09876163257285953, "reward_total_mean": 0.6216350197792053, "reward_meter_mean": 0.8259141445159912, "reward_meter_std": 0.26048386096954346, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9079709053039551, "reward_repeat_soft_std": 0.06270492821931839, "reward_judge_quality_mean": 0.5237500071525574, "reward_judge_quality_std": 0.19078317284584045, "reward_total_composite_mean": 0.6216350197792053, "reward_total_composite_std": 0.1506887674331665} {"timestamp_utc": "2026-04-13T10:59:11Z", "mode": "train", "global_step": 1480, "epoch": 0.1486690105474636, "loss": 0.004, "grad_norm": 8.53366470336914, "learning_rate": 5.518181818181818e-06, "num_tokens": 2613336.0, "completions/mean_length": 94.5, "completions/min_length": 81.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.5, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.8894447088241577, "rewards/meter/std": 0.139001727104187, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8424639701843262, "rewards/repeat_soft/std": 0.075047567486763, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5691879987716675, "rewards/total_composite/std": 0.03418510779738426, "reward": 0.5691879987716675, "reward_std": 0.03418509662151337, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09812415391206741, "sampling/sampling_logp_difference/max": 2.183708667755127, "sampling/importance_sampling_ratio/min": 0.11262307316064835, "sampling/importance_sampling_ratio/mean": 1.0124567747116089, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5634704940021038, "clip_ratio/low_mean": 0.031258889473974705, "clip_ratio/low_min": 0.031258889473974705, "clip_ratio/high_mean": 0.062191106379032135, "clip_ratio/high_max": 0.062191106379032135, "clip_ratio/region_mean": 0.09344999585300684, "reward_total_mean": 0.5691879987716675, "reward_meter_mean": 0.8894447088241577, "reward_meter_std": 0.139001727104187, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8424639701843262, "reward_repeat_soft_std": 0.075047567486763, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5691879987716675, "reward_total_composite_std": 0.03418510779738426} {"timestamp_utc": "2026-04-13T10:59:18Z", "mode": "train", "global_step": 1481, "epoch": 0.14876946258161727, "loss": 0.0185, "grad_norm": 5.042724132537842, "learning_rate": 5.515151515151515e-06, "num_tokens": 2615841.0, "completions/mean_length": 120.125, "completions/min_length": 91.0, "completions/max_length": 142.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 120.125, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 142.0, "rewards/meter/mean": 0.9654009342193604, "rewards/meter/std": 0.02672361209988594, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7379403114318848, "rewards/repeat_soft/std": 0.07007953524589539, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.5003182888031006, "rewards/total_composite/std": 0.07372362166643143, "reward": 0.5003182888031006, "reward_std": 0.07372362166643143, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08715896308422089, "sampling/sampling_logp_difference/max": 2.041287660598755, "sampling/importance_sampling_ratio/min": 0.12986138463020325, "sampling/importance_sampling_ratio/mean": 1.0069060325622559, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5354104712605476, "clip_ratio/low_mean": 0.018736762925982475, "clip_ratio/low_min": 0.018736762925982475, "clip_ratio/high_mean": 0.05698342062532902, "clip_ratio/high_max": 0.05698342062532902, "clip_ratio/region_mean": 0.07572018355131149, "reward_total_mean": 0.5003182888031006, "reward_meter_mean": 0.9654009342193604, "reward_meter_std": 0.02672361209988594, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7379403114318848, "reward_repeat_soft_std": 0.07007953524589539, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.5003182888031006, "reward_total_composite_std": 0.07372362166643143} {"timestamp_utc": "2026-04-13T10:59:24Z", "mode": "train", "global_step": 1482, "epoch": 0.14886991461577098, "loss": 0.0199, "grad_norm": 13.904169082641602, "learning_rate": 5.512121212121213e-06, "num_tokens": 2617555.0, "completions/mean_length": 51.25, "completions/min_length": 46.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.25, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.7591448426246643, "rewards/meter/std": 0.43144935369491577, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9451721906661987, "rewards/repeat_soft/std": 0.06980152428150177, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.549022376537323, "rewards/total_composite/std": 0.11372621357440948, "reward": 0.549022376537323, "reward_std": 0.11372620612382889, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11626243591308594, "sampling/sampling_logp_difference/max": 2.5900814533233643, "sampling/importance_sampling_ratio/min": 0.07501393556594849, "sampling/importance_sampling_ratio/mean": 0.9951414465904236, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6460620984435081, "clip_ratio/low_mean": 0.04374057427048683, "clip_ratio/low_min": 0.04374057427048683, "clip_ratio/high_mean": 0.07585607282817364, "clip_ratio/high_max": 0.07585607282817364, "clip_ratio/region_mean": 0.11959664709866047, "reward_total_mean": 0.549022376537323, "reward_meter_mean": 0.7591448426246643, "reward_meter_std": 0.43144935369491577, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9451721906661987, "reward_repeat_soft_std": 0.06980152428150177, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.549022376537323, "reward_total_composite_std": 0.11372621357440948} {"timestamp_utc": "2026-04-13T10:59:30Z", "mode": "train", "global_step": 1483, "epoch": 0.14897036664992466, "loss": -0.0115, "grad_norm": 16.531383514404297, "learning_rate": 5.50909090909091e-06, "num_tokens": 2618989.0, "completions/mean_length": 26.25, "completions/min_length": 21.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.25, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.6387969255447388, "rewards/meter/std": 0.4885605573654175, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9475217461585999, "rewards/repeat_soft/std": 0.042364854365587234, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.17136012017726898, "rewards/total_composite/mean": 0.5319289565086365, "rewards/total_composite/std": 0.1341337114572525, "reward": 0.5319289565086365, "reward_std": 0.1341337114572525, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12180305272340775, "sampling/sampling_logp_difference/max": 0.840731143951416, "sampling/importance_sampling_ratio/min": 0.431395024061203, "sampling/importance_sampling_ratio/mean": 1.0240724086761475, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9381069242954254, "clip_ratio/low_mean": 0.05408653989434242, "clip_ratio/low_min": 0.05408653989434242, "clip_ratio/high_mean": 0.07852909667417407, "clip_ratio/high_max": 0.07852909667417407, "clip_ratio/region_mean": 0.1326156365685165, "reward_total_mean": 0.5319289565086365, "reward_meter_mean": 0.6387969255447388, "reward_meter_std": 0.4885605573654175, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9475217461585999, "reward_repeat_soft_std": 0.042364854365587234, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.17136012017726898, "reward_total_composite_mean": 0.5319289565086365, "reward_total_composite_std": 0.1341337114572525} {"timestamp_utc": "2026-04-13T10:59:37Z", "mode": "train", "global_step": 1484, "epoch": 0.14907081868407834, "loss": -0.0041, "grad_norm": 8.879104614257812, "learning_rate": 5.506060606060607e-06, "num_tokens": 2620724.0, "completions/mean_length": 59.875, "completions/min_length": 48.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.875, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9281506538391113, "rewards/meter/std": 0.12714247405529022, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9748289585113525, "rewards/repeat_soft/std": 0.026928458362817764, "rewards/judge_quality/mean": 0.5012500286102295, "rewards/judge_quality/std": 0.16974246501922607, "rewards/total_composite/mean": 0.6508640050888062, "rewards/total_composite/std": 0.1229616180062294, "reward": 0.6508640050888062, "reward_std": 0.1229616105556488, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1533975899219513, "sampling/sampling_logp_difference/max": 1.600853443145752, "sampling/importance_sampling_ratio/min": 0.20172427594661713, "sampling/importance_sampling_ratio/mean": 1.0146669149398804, "sampling/importance_sampling_ratio/max": 1.929390549659729, "entropy": 1.1840372160077095, "clip_ratio/low_mean": 0.1217160364612937, "clip_ratio/low_min": 0.1217160364612937, "clip_ratio/high_mean": 0.012295082211494446, "clip_ratio/high_max": 0.012295082211494446, "clip_ratio/region_mean": 0.13401111867278814, "reward_total_mean": 0.6508640050888062, "reward_meter_mean": 0.9281506538391113, "reward_meter_std": 0.12714247405529022, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9748289585113525, "reward_repeat_soft_std": 0.026928458362817764, "reward_judge_quality_mean": 0.5012500286102295, "reward_judge_quality_std": 0.16974246501922607, "reward_total_composite_mean": 0.6508640050888062, "reward_total_composite_std": 0.1229616180062294} {"timestamp_utc": "2026-04-13T10:59:44Z", "mode": "train", "global_step": 1485, "epoch": 0.14917127071823205, "loss": 0.0026, "grad_norm": 8.277298927307129, "learning_rate": 5.5030303030303034e-06, "num_tokens": 2623061.0, "completions/mean_length": 108.125, "completions/min_length": 99.0, "completions/max_length": 122.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.125, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 122.0, "rewards/meter/mean": 0.9773924350738525, "rewards/meter/std": 0.013282565400004387, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9117889404296875, "rewards/repeat_soft/std": 0.04147614538669586, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.6035964488983154, "rewards/total_composite/std": 0.00719041284173727, "reward": 0.6035964488983154, "reward_std": 0.007190412376075983, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11500288546085358, "sampling/sampling_logp_difference/max": 1.3996589183807373, "sampling/importance_sampling_ratio/min": 0.2466810792684555, "sampling/importance_sampling_ratio/mean": 1.0076630115509033, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6722720637917519, "clip_ratio/low_mean": 0.06540508940815926, "clip_ratio/low_min": 0.06540508940815926, "clip_ratio/high_mean": 0.038259128108620644, "clip_ratio/high_max": 0.038259128108620644, "clip_ratio/region_mean": 0.1036642175167799, "reward_total_mean": 0.6035964488983154, "reward_meter_mean": 0.9773924350738525, "reward_meter_std": 0.013282565400004387, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9117889404296875, "reward_repeat_soft_std": 0.04147614538669586, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.6035964488983154, "reward_total_composite_std": 0.00719041284173727} {"timestamp_utc": "2026-04-13T10:59:50Z", "mode": "train", "global_step": 1486, "epoch": 0.14927172275238573, "loss": -0.0171, "grad_norm": 10.497859954833984, "learning_rate": 5.500000000000001e-06, "num_tokens": 2624788.0, "completions/mean_length": 65.875, "completions/min_length": 47.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.875, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.807790994644165, "rewards/meter/std": 0.20373433828353882, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.970165491104126, "rewards/repeat_soft/std": 0.036199379712343216, "rewards/judge_quality/mean": 0.35499998927116394, "rewards/judge_quality/std": 0.11928357183933258, "rewards/total_composite/mean": 0.48034781217575073, "rewards/total_composite/std": 0.20813879370689392, "reward": 0.48034781217575073, "reward_std": 0.20813877880573273, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13982558250427246, "sampling/sampling_logp_difference/max": 1.3635053634643555, "sampling/importance_sampling_ratio/min": 0.25576266646385193, "sampling/importance_sampling_ratio/mean": 0.996860921382904, "sampling/importance_sampling_ratio/max": 1.9622875452041626, "entropy": 1.1020059809088707, "clip_ratio/low_mean": 0.03239944111555815, "clip_ratio/low_min": 0.03239944111555815, "clip_ratio/high_mean": 0.09075272921472788, "clip_ratio/high_max": 0.09075272921472788, "clip_ratio/region_mean": 0.12315217033028603, "reward_total_mean": 0.48034781217575073, "reward_meter_mean": 0.807790994644165, "reward_meter_std": 0.20373433828353882, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.970165491104126, "reward_repeat_soft_std": 0.036199379712343216, "reward_judge_quality_mean": 0.35499998927116394, "reward_judge_quality_std": 0.11928357183933258, "reward_total_composite_mean": 0.48034781217575073, "reward_total_composite_std": 0.20813879370689392} {"timestamp_utc": "2026-04-13T10:59:58Z", "mode": "train", "global_step": 1487, "epoch": 0.14937217478653944, "loss": 0.0495, "grad_norm": 5.990231990814209, "learning_rate": 5.496969696969697e-06, "num_tokens": 2627478.0, "completions/mean_length": 146.25, "completions/min_length": 131.0, "completions/max_length": 159.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 146.25, "completions/min_terminated_length": 131.0, "completions/max_terminated_length": 159.0, "rewards/meter/mean": 0.9197298288345337, "rewards/meter/std": 0.11296083778142929, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.875533401966095, "rewards/repeat_soft/std": 0.05981547012925148, "rewards/judge_quality/mean": 0.5325000286102295, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.6126554012298584, "rewards/total_composite/std": 0.09052237123250961, "reward": 0.6126554012298584, "reward_std": 0.09052237123250961, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11949143558740616, "sampling/sampling_logp_difference/max": 1.4177286624908447, "sampling/importance_sampling_ratio/min": 0.24701178073883057, "sampling/importance_sampling_ratio/mean": 1.0083190202713013, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7267304733395576, "clip_ratio/low_mean": 0.06900496501475573, "clip_ratio/low_min": 0.06900496501475573, "clip_ratio/high_mean": 0.05193363409489393, "clip_ratio/high_max": 0.05193363409489393, "clip_ratio/region_mean": 0.12093859910964966, "reward_total_mean": 0.6126554012298584, "reward_meter_mean": 0.9197298288345337, "reward_meter_std": 0.11296083778142929, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.875533401966095, "reward_repeat_soft_std": 0.05981547012925148, "reward_judge_quality_mean": 0.5325000286102295, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.6126554012298584, "reward_total_composite_std": 0.09052237123250961} {"timestamp_utc": "2026-04-13T11:00:05Z", "mode": "train", "global_step": 1488, "epoch": 0.14947262682069312, "loss": -0.0701, "grad_norm": 15.173059463500977, "learning_rate": 5.493939393939395e-06, "num_tokens": 2628993.0, "completions/mean_length": 30.375, "completions/min_length": 28.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.375, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.8411796689033508, "rewards/meter/std": 0.26568955183029175, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8814701437950134, "rewards/repeat_soft/std": 0.09990499913692474, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.10507651418447495, "rewards/total_composite/mean": 0.5878442525863647, "rewards/total_composite/std": 0.11131254583597183, "reward": 0.5878442525863647, "reward_std": 0.11131253838539124, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09718216955661774, "sampling/sampling_logp_difference/max": 2.024062156677246, "sampling/importance_sampling_ratio/min": 0.13211768865585327, "sampling/importance_sampling_ratio/mean": 0.9881425499916077, "sampling/importance_sampling_ratio/max": 1.7737125158309937, "entropy": 0.5191093683242798, "clip_ratio/low_mean": 0.03064450016245246, "clip_ratio/low_min": 0.03064450016245246, "clip_ratio/high_mean": 0.04625057056546211, "clip_ratio/high_max": 0.04625057056546211, "clip_ratio/region_mean": 0.07689507072791457, "reward_total_mean": 0.5878442525863647, "reward_meter_mean": 0.8411796689033508, "reward_meter_std": 0.26568955183029175, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8814701437950134, "reward_repeat_soft_std": 0.09990499913692474, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.10507651418447495, "reward_total_composite_mean": 0.5878442525863647, "reward_total_composite_std": 0.11131254583597183} {"timestamp_utc": "2026-04-13T11:00:12Z", "mode": "train", "global_step": 1489, "epoch": 0.1495730788548468, "loss": 0.0822, "grad_norm": 9.129810333251953, "learning_rate": 5.490909090909091e-06, "num_tokens": 2630972.0, "completions/mean_length": 82.375, "completions/min_length": 67.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.375, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.8473856449127197, "rewards/meter/std": 0.31352749466896057, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9352773427963257, "rewards/repeat_soft/std": 0.04949399083852768, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.22403764724731445, "rewards/total_composite/mean": 0.5794332027435303, "rewards/total_composite/std": 0.13291306793689728, "reward": 0.5794332027435303, "reward_std": 0.13291305303573608, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11948537081480026, "sampling/sampling_logp_difference/max": 1.7878718376159668, "sampling/importance_sampling_ratio/min": 0.16731585562229156, "sampling/importance_sampling_ratio/mean": 1.0016365051269531, "sampling/importance_sampling_ratio/max": 1.9319500923156738, "entropy": 0.756651908159256, "clip_ratio/low_mean": 0.029689370654523373, "clip_ratio/low_min": 0.029689370654523373, "clip_ratio/high_mean": 0.10294715873897076, "clip_ratio/high_max": 0.10294715873897076, "clip_ratio/region_mean": 0.13263652939349413, "reward_total_mean": 0.5794332027435303, "reward_meter_mean": 0.8473856449127197, "reward_meter_std": 0.31352749466896057, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9352773427963257, "reward_repeat_soft_std": 0.04949399083852768, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.22403764724731445, "reward_total_composite_mean": 0.5794332027435303, "reward_total_composite_std": 0.13291306793689728} {"timestamp_utc": "2026-04-13T11:00:18Z", "mode": "train", "global_step": 1490, "epoch": 0.1496735308890005, "loss": 0.0197, "grad_norm": 8.857263565063477, "learning_rate": 5.487878787878789e-06, "num_tokens": 2632388.0, "completions/mean_length": 23.0, "completions/min_length": 23.0, "completions/max_length": 23.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.0, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 23.0, "rewards/meter/mean": 0.9874359369277954, "rewards/meter/std": 0.006660922896116972, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9256622195243835, "rewards/repeat_soft/std": 0.049346212297677994, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.6204668283462524, "rewards/total_composite/std": 0.009920109063386917, "reward": 0.6204668283462524, "reward_std": 0.009920092299580574, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07062021642923355, "sampling/sampling_logp_difference/max": 1.7588739395141602, "sampling/importance_sampling_ratio/min": 0.17223870754241943, "sampling/importance_sampling_ratio/mean": 1.0116403102874756, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3152022659778595, "clip_ratio/low_mean": 0.027173913549631834, "clip_ratio/low_min": 0.027173913549631834, "clip_ratio/high_mean": 0.021739130839705467, "clip_ratio/high_max": 0.021739130839705467, "clip_ratio/region_mean": 0.0489130443893373, "reward_total_mean": 0.6204668283462524, "reward_meter_mean": 0.9874359369277954, "reward_meter_std": 0.006660922896116972, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9256622195243835, "reward_repeat_soft_std": 0.049346212297677994, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.6204668283462524, "reward_total_composite_std": 0.009920109063386917} {"timestamp_utc": "2026-04-13T11:00:24Z", "mode": "train", "global_step": 1491, "epoch": 0.1497739829231542, "loss": 0.005, "grad_norm": 21.160472869873047, "learning_rate": 5.484848484848485e-06, "num_tokens": 2633915.0, "completions/mean_length": 22.875, "completions/min_length": 18.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.875, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.9593546390533447, "rewards/meter/std": 0.03311125934123993, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9424046277999878, "rewards/repeat_soft/std": 0.049576517194509506, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.0843462198972702, "rewards/total_composite/mean": 0.5809230804443359, "rewards/total_composite/std": 0.056494951248168945, "reward": 0.5809230804443359, "reward_std": 0.05649494379758835, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13785071671009064, "sampling/sampling_logp_difference/max": 1.004812240600586, "sampling/importance_sampling_ratio/min": 0.3661133646965027, "sampling/importance_sampling_ratio/mean": 1.0086857080459595, "sampling/importance_sampling_ratio/max": 1.6974269151687622, "entropy": 0.9706518948078156, "clip_ratio/low_mean": 0.015567766036838293, "clip_ratio/low_min": 0.015567766036838293, "clip_ratio/high_mean": 0.08649774361401796, "clip_ratio/high_max": 0.08649774361401796, "clip_ratio/region_mean": 0.10206550965085626, "reward_total_mean": 0.5809230804443359, "reward_meter_mean": 0.9593546390533447, "reward_meter_std": 0.03311125934123993, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9424046277999878, "reward_repeat_soft_std": 0.049576517194509506, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.0843462198972702, "reward_total_composite_mean": 0.5809230804443359, "reward_total_composite_std": 0.056494951248168945} {"timestamp_utc": "2026-04-13T11:00:35Z", "mode": "train", "global_step": 1492, "epoch": 0.14987443495730787, "loss": -0.1434, "grad_norm": 2.424750804901123, "learning_rate": 5.4818181818181825e-06, "num_tokens": 2635684.0, "completions/mean_length": 113.125, "completions/min_length": 49.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 56.142860412597656, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9021384716033936, "rewards/meter/std": 0.15267567336559296, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9634636640548706, "rewards/repeat_soft/std": 0.030924834311008453, "rewards/judge_quality/mean": 0.38874998688697815, "rewards/judge_quality/std": 0.13767844438552856, "rewards/total_composite/mean": 0.5216336250305176, "rewards/total_composite/std": 0.21433673799037933, "reward": 0.5216336250305176, "reward_std": 0.21433670818805695, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12702292203903198, "sampling/sampling_logp_difference/max": 1.3553638458251953, "sampling/importance_sampling_ratio/min": 0.2578534483909607, "sampling/importance_sampling_ratio/mean": 0.9867607355117798, "sampling/importance_sampling_ratio/max": 1.7102025747299194, "entropy": 0.7838902771472931, "clip_ratio/low_mean": 0.004464285913854837, "clip_ratio/low_min": 0.004464285913854837, "clip_ratio/high_mean": 0.10617516562342644, "clip_ratio/high_max": 0.10617516562342644, "clip_ratio/region_mean": 0.11063945153728127, "reward_total_mean": 0.5216336250305176, "reward_meter_mean": 0.9021384716033936, "reward_meter_std": 0.15267567336559296, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9634636640548706, "reward_repeat_soft_std": 0.030924834311008453, "reward_judge_quality_mean": 0.38874998688697815, "reward_judge_quality_std": 0.13767844438552856, "reward_total_composite_mean": 0.5216336250305176, "reward_total_composite_std": 0.21433673799037933} {"timestamp_utc": "2026-04-13T11:00:42Z", "mode": "train", "global_step": 1493, "epoch": 0.14997488699146158, "loss": -0.0612, "grad_norm": 8.406460762023926, "learning_rate": 5.478787878787879e-06, "num_tokens": 2637724.0, "completions/mean_length": 87.0, "completions/min_length": 78.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.0, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.9535136222839355, "rewards/meter/std": 0.03741809353232384, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8257770538330078, "rewards/repeat_soft/std": 0.12242306768894196, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.21357084810733795, "rewards/total_composite/mean": 0.634614884853363, "rewards/total_composite/std": 0.1459803432226181, "reward": 0.634614884853363, "reward_std": 0.1459803283214569, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10829455405473709, "sampling/sampling_logp_difference/max": 1.737403392791748, "sampling/importance_sampling_ratio/min": 0.17597675323486328, "sampling/importance_sampling_ratio/mean": 1.0054004192352295, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6760761961340904, "clip_ratio/low_mean": 0.07716662669554353, "clip_ratio/low_min": 0.07716662669554353, "clip_ratio/high_mean": 0.028583532199263573, "clip_ratio/high_max": 0.028583532199263573, "clip_ratio/region_mean": 0.1057501588948071, "reward_total_mean": 0.634614884853363, "reward_meter_mean": 0.9535136222839355, "reward_meter_std": 0.03741809353232384, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8257770538330078, "reward_repeat_soft_std": 0.12242306768894196, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.21357084810733795, "reward_total_composite_mean": 0.634614884853363, "reward_total_composite_std": 0.1459803432226181} {"timestamp_utc": "2026-04-13T11:00:48Z", "mode": "train", "global_step": 1494, "epoch": 0.15007533902561526, "loss": 0.0655, "grad_norm": 11.62306022644043, "learning_rate": 5.475757575757576e-06, "num_tokens": 2639318.0, "completions/mean_length": 46.25, "completions/min_length": 36.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.25, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.7219693660736084, "rewards/meter/std": 0.33736079931259155, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9877862930297852, "rewards/repeat_soft/std": 0.009149615652859211, "rewards/judge_quality/mean": 0.5724999904632568, "rewards/judge_quality/std": 0.194770947098732, "rewards/total_composite/mean": 0.6372451186180115, "rewards/total_composite/std": 0.19359150528907776, "reward": 0.6372451186180115, "reward_std": 0.19359152019023895, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12142278254032135, "sampling/sampling_logp_difference/max": 1.2413525581359863, "sampling/importance_sampling_ratio/min": 0.28899309039115906, "sampling/importance_sampling_ratio/mean": 0.9877916574478149, "sampling/importance_sampling_ratio/max": 1.7015717029571533, "entropy": 0.7238209135830402, "clip_ratio/low_mean": 0.095611322671175, "clip_ratio/low_min": 0.095611322671175, "clip_ratio/high_mean": 0.03800287377089262, "clip_ratio/high_max": 0.03800287377089262, "clip_ratio/region_mean": 0.13361419644206762, "reward_total_mean": 0.6372451186180115, "reward_meter_mean": 0.7219693660736084, "reward_meter_std": 0.33736079931259155, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9877862930297852, "reward_repeat_soft_std": 0.009149615652859211, "reward_judge_quality_mean": 0.5724999904632568, "reward_judge_quality_std": 0.194770947098732, "reward_total_composite_mean": 0.6372451186180115, "reward_total_composite_std": 0.19359150528907776} {"timestamp_utc": "2026-04-13T11:00:55Z", "mode": "train", "global_step": 1495, "epoch": 0.15017579105976897, "loss": 0.0637, "grad_norm": 9.856406211853027, "learning_rate": 5.472727272727273e-06, "num_tokens": 2641005.0, "completions/mean_length": 58.875, "completions/min_length": 53.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.875, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.6126219034194946, "rewards/meter/std": 0.4486890733242035, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9383866786956787, "rewards/repeat_soft/std": 0.04639485850930214, "rewards/judge_quality/mean": 0.41749998927116394, "rewards/judge_quality/std": 0.06902380287647247, "rewards/total_composite/mean": 0.4990963935852051, "rewards/total_composite/std": 0.11686215549707413, "reward": 0.4990963935852051, "reward_std": 0.11686215549707413, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11863676458597183, "sampling/sampling_logp_difference/max": 1.8329658508300781, "sampling/importance_sampling_ratio/min": 0.1599385142326355, "sampling/importance_sampling_ratio/mean": 1.023113489151001, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7165866494178772, "clip_ratio/low_mean": 0.05935907084494829, "clip_ratio/low_min": 0.05935907084494829, "clip_ratio/high_mean": 0.041931509505957365, "clip_ratio/high_max": 0.041931509505957365, "clip_ratio/region_mean": 0.10129058035090566, "reward_total_mean": 0.4990963935852051, "reward_meter_mean": 0.6126219034194946, "reward_meter_std": 0.4486890733242035, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9383866786956787, "reward_repeat_soft_std": 0.04639485850930214, "reward_judge_quality_mean": 0.41749998927116394, "reward_judge_quality_std": 0.06902380287647247, "reward_total_composite_mean": 0.4990963935852051, "reward_total_composite_std": 0.11686215549707413} {"timestamp_utc": "2026-04-13T11:01:08Z", "mode": "train", "global_step": 1496, "epoch": 0.15027624309392265, "loss": 0.0202, "grad_norm": 13.381893157958984, "learning_rate": 5.469696969696971e-06, "num_tokens": 2642740.0, "completions/mean_length": 44.875, "completions/min_length": 35.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.875, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.8217130899429321, "rewards/meter/std": 0.16472452878952026, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9239823818206787, "rewards/repeat_soft/std": 0.07076884806156158, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.5703728199005127, "rewards/total_composite/std": 0.05117390304803848, "reward": 0.5703728199005127, "reward_std": 0.05117391049861908, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1393130123615265, "sampling/sampling_logp_difference/max": 1.482508659362793, "sampling/importance_sampling_ratio/min": 0.22706735134124756, "sampling/importance_sampling_ratio/mean": 0.9985095262527466, "sampling/importance_sampling_ratio/max": 1.9565730094909668, "entropy": 1.0107583776116371, "clip_ratio/low_mean": 0.03506616735830903, "clip_ratio/low_min": 0.03506616735830903, "clip_ratio/high_mean": 0.11435185372829437, "clip_ratio/high_max": 0.11435185372829437, "clip_ratio/region_mean": 0.1494180210866034, "reward_total_mean": 0.5703728199005127, "reward_meter_mean": 0.8217130899429321, "reward_meter_std": 0.16472452878952026, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9239823818206787, "reward_repeat_soft_std": 0.07076884806156158, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.5703728199005127, "reward_total_composite_std": 0.05117390304803848} {"timestamp_utc": "2026-04-13T11:01:14Z", "mode": "train", "global_step": 1497, "epoch": 0.15037669512807633, "loss": -0.0586, "grad_norm": 15.628218650817871, "learning_rate": 5.466666666666667e-06, "num_tokens": 2644152.0, "completions/mean_length": 27.5, "completions/min_length": 22.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.5, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.8665649890899658, "rewards/meter/std": 0.3190095126628876, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9560876488685608, "rewards/repeat_soft/std": 0.018136821687221527, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.5825812816619873, "rewards/total_composite/std": 0.09054209291934967, "reward": 0.5825812816619873, "reward_std": 0.09054208546876907, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14445458352565765, "sampling/sampling_logp_difference/max": 1.2789459228515625, "sampling/importance_sampling_ratio/min": 0.278330534696579, "sampling/importance_sampling_ratio/mean": 1.0274255275726318, "sampling/importance_sampling_ratio/max": 1.9164459705352783, "entropy": 1.5635968893766403, "clip_ratio/low_mean": 0.03804347664117813, "clip_ratio/low_min": 0.03804347664117813, "clip_ratio/high_mean": 0.14897077903151512, "clip_ratio/high_max": 0.14897077903151512, "clip_ratio/region_mean": 0.18701425567269325, "reward_total_mean": 0.5825812816619873, "reward_meter_mean": 0.8665649890899658, "reward_meter_std": 0.3190095126628876, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9560876488685608, "reward_repeat_soft_std": 0.018136821687221527, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.5825812816619873, "reward_total_composite_std": 0.09054209291934967} {"timestamp_utc": "2026-04-13T11:01:20Z", "mode": "train", "global_step": 1498, "epoch": 0.15047714716223004, "loss": 0.0737, "grad_norm": 9.467653274536133, "learning_rate": 5.463636363636364e-06, "num_tokens": 2645644.0, "completions/mean_length": 44.5, "completions/min_length": 37.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.5, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.8815276622772217, "rewards/meter/std": 0.28615087270736694, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9526903629302979, "rewards/repeat_soft/std": 0.04133959859609604, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.6274303197860718, "rewards/total_composite/std": 0.1451360434293747, "reward": 0.6274303197860718, "reward_std": 0.1451360434293747, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14966563880443573, "sampling/sampling_logp_difference/max": 2.6682653427124023, "sampling/importance_sampling_ratio/min": 0.06937245279550552, "sampling/importance_sampling_ratio/mean": 0.9882627725601196, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.726091668009758, "clip_ratio/low_mean": 0.07268185913562775, "clip_ratio/low_min": 0.07268185913562775, "clip_ratio/high_mean": 0.0625273548066616, "clip_ratio/high_max": 0.0625273548066616, "clip_ratio/region_mean": 0.13520921394228935, "reward_total_mean": 0.6274303197860718, "reward_meter_mean": 0.8815276622772217, "reward_meter_std": 0.28615087270736694, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9526903629302979, "reward_repeat_soft_std": 0.04133959859609604, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.6274303197860718, "reward_total_composite_std": 0.1451360434293747} {"timestamp_utc": "2026-04-13T11:01:27Z", "mode": "train", "global_step": 1499, "epoch": 0.15057759919638372, "loss": -0.0227, "grad_norm": 8.826386451721191, "learning_rate": 5.460606060606061e-06, "num_tokens": 2647574.0, "completions/mean_length": 78.25, "completions/min_length": 60.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.25, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.5142233371734619, "rewards/meter/std": 0.3521144390106201, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9376263618469238, "rewards/repeat_soft/std": 0.07676321268081665, "rewards/judge_quality/mean": 0.41874998807907104, "rewards/judge_quality/std": 0.14574317634105682, "rewards/total_composite/mean": 0.47821107506752014, "rewards/total_composite/std": 0.0897725373506546, "reward": 0.47821107506752014, "reward_std": 0.089772529900074, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1335970014333725, "sampling/sampling_logp_difference/max": 1.471923828125, "sampling/importance_sampling_ratio/min": 0.22948358952999115, "sampling/importance_sampling_ratio/mean": 1.028091549873352, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.117735967040062, "clip_ratio/low_mean": 0.08763296157121658, "clip_ratio/low_min": 0.08763296157121658, "clip_ratio/high_mean": 0.04597425367683172, "clip_ratio/high_max": 0.04597425367683172, "clip_ratio/region_mean": 0.1336072152480483, "reward_total_mean": 0.47821107506752014, "reward_meter_mean": 0.5142233371734619, "reward_meter_std": 0.3521144390106201, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9376263618469238, "reward_repeat_soft_std": 0.07676321268081665, "reward_judge_quality_mean": 0.41874998807907104, "reward_judge_quality_std": 0.14574317634105682, "reward_total_composite_mean": 0.47821107506752014, "reward_total_composite_std": 0.0897725373506546} {"timestamp_utc": "2026-04-13T11:01:34Z", "mode": "train", "global_step": 1500, "epoch": 0.15067805123053743, "loss": 0.0307, "grad_norm": 10.32536506652832, "learning_rate": 5.457575757575758e-06, "num_tokens": 2649204.0, "completions/mean_length": 51.75, "completions/min_length": 45.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.75, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.7600724697113037, "rewards/meter/std": 0.37609532475471497, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9694873690605164, "rewards/repeat_soft/std": 0.04620639607310295, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.6760584115982056, "rewards/total_composite/std": 0.216745987534523, "reward": 0.6760584115982056, "reward_std": 0.216745987534523, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15095128118991852, "sampling/sampling_logp_difference/max": 4.3736796379089355, "sampling/importance_sampling_ratio/min": 0.012604773975908756, "sampling/importance_sampling_ratio/mean": 1.0109680891036987, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8784070312976837, "clip_ratio/low_mean": 0.0693869749084115, "clip_ratio/low_min": 0.0693869749084115, "clip_ratio/high_mean": 0.0614222576841712, "clip_ratio/high_max": 0.0614222576841712, "clip_ratio/region_mean": 0.1308092325925827, "reward_total_mean": 0.6760584115982056, "reward_meter_mean": 0.7600724697113037, "reward_meter_std": 0.37609532475471497, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9694873690605164, "reward_repeat_soft_std": 0.04620639607310295, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.6760584115982056, "reward_total_composite_std": 0.216745987534523} {"timestamp_utc": "2026-04-13T11:02:19Z", "mode": "eval", "global_step": 1500, "epoch": 0.15067805123053743, "eval_loss": NaN, "eval_runtime": 44.9802, "eval_samples_per_second": 1.779, "eval_steps_per_second": 0.222, "eval_num_tokens": 2649204.0, "eval_completions/mean_length": 85.5875, "eval_completions/min_length": 32.7, "eval_completions/max_length": 162.7, "eval_completions/clipped_ratio": 0.0125, "eval_completions/mean_terminated_length": 80.42500076293945, "eval_completions/min_terminated_length": 32.7, "eval_completions/max_terminated_length": 127.9, "eval_rewards/meter/mean": 0.8153466045856476, "eval_rewards/meter/std": 0.2554396173916757, "eval_rewards/count_adherence/mean": 0.9520833134651184, "eval_rewards/count_adherence/std": 0.07691999115049838, "eval_rewards/hard_gate/mean": 0.975, "eval_rewards/hard_gate/std": 0.07071067690849304, "eval_rewards/repeat_soft/mean": 0.9100002408027649, "eval_rewards/repeat_soft/std": 0.07414132617413997, "eval_rewards/judge_quality/mean": 0.49362499713897706, "eval_rewards/judge_quality/std": 0.16515217600390314, "eval_rewards/total_composite/mean": 0.582668250799179, "eval_rewards/total_composite/std": 0.15843873098492622, "eval_reward": 0.582668250799179, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.06454889848828316, "eval_sampling/sampling_logp_difference/max": 1.154489517211914, "eval_sampling/importance_sampling_ratio/min": 0.3234517902135849, "eval_sampling/importance_sampling_ratio/mean": 1.0130776166915894, "eval_sampling/importance_sampling_ratio/max": 1.4355666995048524, "eval_entropy": 0.7258538126945495, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.582668250799179, "eval_reward_meter_mean": 0.8153466045856476, "eval_reward_meter_std": 0.2554396173916757, "eval_reward_count_adherence_mean": 0.9520833134651184, "eval_reward_count_adherence_std": 0.07691999115049838, "eval_reward_hard_gate_mean": 0.975, "eval_reward_hard_gate_std": 0.07071067690849304, "eval_reward_repeat_soft_mean": 0.9100002408027649, "eval_reward_repeat_soft_std": 0.07414132617413997, "eval_reward_judge_quality_mean": 0.49362499713897706, "eval_reward_judge_quality_std": 0.16515217600390314, "eval_reward_total_composite_mean": 0.582668250799179, "eval_reward_total_composite_std": 0.15843873098492622} {"timestamp_utc": "2026-04-13T11:02:34Z", "mode": "train", "global_step": 1501, "epoch": 0.1507785032646911, "loss": -0.2061, "grad_norm": 2.157881498336792, "learning_rate": 5.4545454545454545e-06, "num_tokens": 2651603.0, "completions/mean_length": 162.875, "completions/min_length": 101.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 113.00000762939453, "completions/min_terminated_length": 101.0, "completions/max_terminated_length": 127.0, "rewards/meter/mean": 0.8726204633712769, "rewards/meter/std": 0.1572498232126236, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.1414213627576828, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8686163425445557, "rewards/repeat_soft/std": 0.08269593119621277, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.13593590259552002, "rewards/total_composite/mean": 0.44510364532470703, "rewards/total_composite/std": 0.18498866260051727, "reward": 0.44510364532470703, "reward_std": 0.18498864769935608, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10576401650905609, "sampling/sampling_logp_difference/max": 1.4901682138442993, "sampling/importance_sampling_ratio/min": 0.2253347635269165, "sampling/importance_sampling_ratio/mean": 1.0134966373443604, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6739145219326019, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09370322432368994, "clip_ratio/high_max": 0.09370322432368994, "clip_ratio/region_mean": 0.09370322432368994, "reward_total_mean": 0.44510364532470703, "reward_meter_mean": 0.8726204633712769, "reward_meter_std": 0.1572498232126236, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.1414213627576828, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8686163425445557, "reward_repeat_soft_std": 0.08269593119621277, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.13593590259552002, "reward_total_composite_mean": 0.44510364532470703, "reward_total_composite_std": 0.18498866260051727} {"timestamp_utc": "2026-04-13T11:02:40Z", "mode": "train", "global_step": 1502, "epoch": 0.1508789552988448, "loss": 0.0589, "grad_norm": 19.415096282958984, "learning_rate": 5.451515151515152e-06, "num_tokens": 2653263.0, "completions/mean_length": 49.5, "completions/min_length": 42.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.5, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.8572096824645996, "rewards/meter/std": 0.30716466903686523, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8570389151573181, "rewards/repeat_soft/std": 0.11212202161550522, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.5695780515670776, "rewards/total_composite/std": 0.08213376253843307, "reward": 0.5695780515670776, "reward_std": 0.08213376253843307, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10910096019506454, "sampling/sampling_logp_difference/max": 1.8397150039672852, "sampling/importance_sampling_ratio/min": 0.15886269509792328, "sampling/importance_sampling_ratio/mean": 1.006670355796814, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6812138892710209, "clip_ratio/low_mean": 0.015625, "clip_ratio/low_min": 0.015625, "clip_ratio/high_mean": 0.10409970721229911, "clip_ratio/high_max": 0.10409970721229911, "clip_ratio/region_mean": 0.11972470721229911, "reward_total_mean": 0.5695780515670776, "reward_meter_mean": 0.8572096824645996, "reward_meter_std": 0.30716466903686523, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8570389151573181, "reward_repeat_soft_std": 0.11212202161550522, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.5695780515670776, "reward_total_composite_std": 0.08213376253843307} {"timestamp_utc": "2026-04-13T11:02:47Z", "mode": "train", "global_step": 1503, "epoch": 0.1509794073329985, "loss": -0.0259, "grad_norm": 8.051602363586426, "learning_rate": 5.448484848484848e-06, "num_tokens": 2655185.0, "completions/mean_length": 60.25, "completions/min_length": 52.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.25, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9878210425376892, "rewards/meter/std": 0.007200425956398249, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9534200429916382, "rewards/repeat_soft/std": 0.04252530634403229, "rewards/judge_quality/mean": 0.4312499761581421, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.6198652386665344, "rewards/total_composite/std": 0.008003455586731434, "reward": 0.6198652386665344, "reward_std": 0.008003444410860538, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1118922159075737, "sampling/sampling_logp_difference/max": 1.7507295608520508, "sampling/importance_sampling_ratio/min": 0.1736472100019455, "sampling/importance_sampling_ratio/mean": 1.0283818244934082, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7226510271430016, "clip_ratio/low_mean": 0.053972672671079636, "clip_ratio/low_min": 0.053972672671079636, "clip_ratio/high_mean": 0.06846752669662237, "clip_ratio/high_max": 0.06846752669662237, "clip_ratio/region_mean": 0.12244019936770201, "reward_total_mean": 0.6198652386665344, "reward_meter_mean": 0.9878210425376892, "reward_meter_std": 0.007200425956398249, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9534200429916382, "reward_repeat_soft_std": 0.04252530634403229, "reward_judge_quality_mean": 0.4312499761581421, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.6198652386665344, "reward_total_composite_std": 0.008003455586731434} {"timestamp_utc": "2026-04-13T11:02:55Z", "mode": "train", "global_step": 1504, "epoch": 0.15107985936715218, "loss": 0.0935, "grad_norm": 5.836754322052002, "learning_rate": 5.445454545454546e-06, "num_tokens": 2657649.0, "completions/mean_length": 137.0, "completions/min_length": 96.0, "completions/max_length": 156.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 137.0, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 156.0, "rewards/meter/mean": 0.8068217039108276, "rewards/meter/std": 0.33071011304855347, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8825508952140808, "rewards/repeat_soft/std": 0.0859769731760025, "rewards/judge_quality/mean": 0.5237500071525574, "rewards/judge_quality/std": 0.19078317284584045, "rewards/total_composite/mean": 0.6047463417053223, "rewards/total_composite/std": 0.14661374688148499, "reward": 0.6047463417053223, "reward_std": 0.14661374688148499, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1181572675704956, "sampling/sampling_logp_difference/max": 2.094663143157959, "sampling/importance_sampling_ratio/min": 0.12311170995235443, "sampling/importance_sampling_ratio/mean": 1.0147782564163208, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7575431764125824, "clip_ratio/low_mean": 0.051041049882769585, "clip_ratio/low_min": 0.051041049882769585, "clip_ratio/high_mean": 0.07136560790240765, "clip_ratio/high_max": 0.07136560790240765, "clip_ratio/region_mean": 0.12240665778517723, "reward_total_mean": 0.6047463417053223, "reward_meter_mean": 0.8068217039108276, "reward_meter_std": 0.33071011304855347, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8825508952140808, "reward_repeat_soft_std": 0.0859769731760025, "reward_judge_quality_mean": 0.5237500071525574, "reward_judge_quality_std": 0.19078317284584045, "reward_total_composite_mean": 0.6047463417053223, "reward_total_composite_std": 0.14661374688148499} {"timestamp_utc": "2026-04-13T11:03:01Z", "mode": "train", "global_step": 1505, "epoch": 0.1511803114013059, "loss": 0.0829, "grad_norm": 16.15242576599121, "learning_rate": 5.442424242424243e-06, "num_tokens": 2659213.0, "completions/mean_length": 34.5, "completions/min_length": 30.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.5, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.9436606764793396, "rewards/meter/std": 0.03392787277698517, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9545820951461792, "rewards/repeat_soft/std": 0.04496331885457039, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.7173162698745728, "rewards/total_composite/std": 0.16378594934940338, "reward": 0.7173162698745728, "reward_std": 0.16378596425056458, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08711379766464233, "sampling/sampling_logp_difference/max": 1.1202850341796875, "sampling/importance_sampling_ratio/min": 0.32618680596351624, "sampling/importance_sampling_ratio/mean": 1.0169613361358643, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.553403202444315, "clip_ratio/low_mean": 0.05818144977092743, "clip_ratio/low_min": 0.05818144977092743, "clip_ratio/high_mean": 0.04139785002917051, "clip_ratio/high_max": 0.04139785002917051, "clip_ratio/region_mean": 0.09957929980009794, "reward_total_mean": 0.7173162698745728, "reward_meter_mean": 0.9436606764793396, "reward_meter_std": 0.03392787277698517, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9545820951461792, "reward_repeat_soft_std": 0.04496331885457039, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.7173162698745728, "reward_total_composite_std": 0.16378594934940338} {"timestamp_utc": "2026-04-13T11:03:08Z", "mode": "train", "global_step": 1506, "epoch": 0.15128076343545957, "loss": 0.0089, "grad_norm": 7.340456008911133, "learning_rate": 5.43939393939394e-06, "num_tokens": 2661594.0, "completions/mean_length": 113.625, "completions/min_length": 102.0, "completions/max_length": 123.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 113.625, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.8746345043182373, "rewards/meter/std": 0.2627345025539398, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8636871576309204, "rewards/repeat_soft/std": 0.07790645956993103, "rewards/judge_quality/mean": 0.44874998927116394, "rewards/judge_quality/std": 0.16137246787548065, "rewards/total_composite/mean": 0.5864561200141907, "rewards/total_composite/std": 0.13236787915229797, "reward": 0.5864561200141907, "reward_std": 0.13236786425113678, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09377837926149368, "sampling/sampling_logp_difference/max": 2.0795183181762695, "sampling/importance_sampling_ratio/min": 0.12499039620161057, "sampling/importance_sampling_ratio/mean": 1.0091928243637085, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5727298632264137, "clip_ratio/low_mean": 0.04311327030882239, "clip_ratio/low_min": 0.04311327030882239, "clip_ratio/high_mean": 0.046685791574418545, "clip_ratio/high_max": 0.046685791574418545, "clip_ratio/region_mean": 0.08979906188324094, "reward_total_mean": 0.5864561200141907, "reward_meter_mean": 0.8746345043182373, "reward_meter_std": 0.2627345025539398, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8636871576309204, "reward_repeat_soft_std": 0.07790645956993103, "reward_judge_quality_mean": 0.44874998927116394, "reward_judge_quality_std": 0.16137246787548065, "reward_total_composite_mean": 0.5864561200141907, "reward_total_composite_std": 0.13236787915229797} {"timestamp_utc": "2026-04-13T11:03:16Z", "mode": "train", "global_step": 1507, "epoch": 0.15138121546961325, "loss": -0.0042, "grad_norm": 7.900362968444824, "learning_rate": 5.436363636363636e-06, "num_tokens": 2663816.0, "completions/mean_length": 99.75, "completions/min_length": 85.0, "completions/max_length": 131.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 99.75, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.7890487909317017, "rewards/meter/std": 0.29200175404548645, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9465805888175964, "rewards/repeat_soft/std": 0.04801928624510765, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5079265236854553, "rewards/total_composite/std": 0.2088804692029953, "reward": 0.5079265236854553, "reward_std": 0.2088804692029953, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1373182237148285, "sampling/sampling_logp_difference/max": 2.233384132385254, "sampling/importance_sampling_ratio/min": 0.1071651503443718, "sampling/importance_sampling_ratio/mean": 0.9960684776306152, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7693092972040176, "clip_ratio/low_mean": 0.016129031777381897, "clip_ratio/low_min": 0.016129031777381897, "clip_ratio/high_mean": 0.11910470807924867, "clip_ratio/high_max": 0.11910470807924867, "clip_ratio/region_mean": 0.13523373985663056, "reward_total_mean": 0.5079265236854553, "reward_meter_mean": 0.7890487909317017, "reward_meter_std": 0.29200175404548645, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9465805888175964, "reward_repeat_soft_std": 0.04801928624510765, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5079265236854553, "reward_total_composite_std": 0.2088804692029953} {"timestamp_utc": "2026-04-13T11:03:22Z", "mode": "train", "global_step": 1508, "epoch": 0.15148166750376696, "loss": 0.0014, "grad_norm": 11.292783737182617, "learning_rate": 5.4333333333333335e-06, "num_tokens": 2665539.0, "completions/mean_length": 46.375, "completions/min_length": 42.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.375, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.948706865310669, "rewards/meter/std": 0.10278742760419846, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8677626848220825, "rewards/repeat_soft/std": 0.03913799673318863, "rewards/judge_quality/mean": 0.48124998807907104, "rewards/judge_quality/std": 0.2294986993074417, "rewards/total_composite/mean": 0.6284198760986328, "rewards/total_composite/std": 0.15320169925689697, "reward": 0.6284198760986328, "reward_std": 0.15320169925689697, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09286518394947052, "sampling/sampling_logp_difference/max": 0.9282870292663574, "sampling/importance_sampling_ratio/min": 0.39523017406463623, "sampling/importance_sampling_ratio/mean": 1.0193724632263184, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6417422741651535, "clip_ratio/low_mean": 0.06929143890738487, "clip_ratio/low_min": 0.06929143890738487, "clip_ratio/high_mean": 0.03787878900766373, "clip_ratio/high_max": 0.03787878900766373, "clip_ratio/region_mean": 0.1071702279150486, "reward_total_mean": 0.6284198760986328, "reward_meter_mean": 0.948706865310669, "reward_meter_std": 0.10278742760419846, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8677626848220825, "reward_repeat_soft_std": 0.03913799673318863, "reward_judge_quality_mean": 0.48124998807907104, "reward_judge_quality_std": 0.2294986993074417, "reward_total_composite_mean": 0.6284198760986328, "reward_total_composite_std": 0.15320169925689697} {"timestamp_utc": "2026-04-13T11:03:29Z", "mode": "train", "global_step": 1509, "epoch": 0.15158211953792064, "loss": 0.0009, "grad_norm": 8.47575855255127, "learning_rate": 5.430303030303032e-06, "num_tokens": 2667852.0, "completions/mean_length": 98.125, "completions/min_length": 92.0, "completions/max_length": 106.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.125, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.6732974052429199, "rewards/meter/std": 0.3062664270401001, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9173442721366882, "rewards/repeat_soft/std": 0.08115822076797485, "rewards/judge_quality/mean": 0.5525000095367432, "rewards/judge_quality/std": 0.18941642343997955, "rewards/total_composite/mean": 0.5861427187919617, "rewards/total_composite/std": 0.1619904786348343, "reward": 0.5861427187919617, "reward_std": 0.1619904637336731, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12370063364505768, "sampling/sampling_logp_difference/max": 2.313128709793091, "sampling/importance_sampling_ratio/min": 0.09895117580890656, "sampling/importance_sampling_ratio/mean": 1.005765438079834, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7182970643043518, "clip_ratio/low_mean": 0.07507901545614004, "clip_ratio/low_min": 0.07507901545614004, "clip_ratio/high_mean": 0.059414125978946686, "clip_ratio/high_max": 0.059414125978946686, "clip_ratio/region_mean": 0.13449314143508673, "reward_total_mean": 0.5861427187919617, "reward_meter_mean": 0.6732974052429199, "reward_meter_std": 0.3062664270401001, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9173442721366882, "reward_repeat_soft_std": 0.08115822076797485, "reward_judge_quality_mean": 0.5525000095367432, "reward_judge_quality_std": 0.18941642343997955, "reward_total_composite_mean": 0.5861427187919617, "reward_total_composite_std": 0.1619904786348343} {"timestamp_utc": "2026-04-13T11:03:35Z", "mode": "train", "global_step": 1510, "epoch": 0.15168257157207435, "loss": 0.0284, "grad_norm": 20.081600189208984, "learning_rate": 5.427272727272728e-06, "num_tokens": 2669263.0, "completions/mean_length": 22.375, "completions/min_length": 19.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.375, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.6335141658782959, "rewards/meter/std": 0.4142892062664032, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9450810551643372, "rewards/repeat_soft/std": 0.02819715067744255, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.5171016454696655, "rewards/total_composite/std": 0.114821657538414, "reward": 0.5171016454696655, "reward_std": 0.11482164263725281, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14239293336868286, "sampling/sampling_logp_difference/max": 1.3075666427612305, "sampling/importance_sampling_ratio/min": 0.27047741413116455, "sampling/importance_sampling_ratio/mean": 1.0207654237747192, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9001251459121704, "clip_ratio/low_mean": 0.0656578941270709, "clip_ratio/low_min": 0.0656578941270709, "clip_ratio/high_mean": 0.09755183570086956, "clip_ratio/high_max": 0.09755183570086956, "clip_ratio/region_mean": 0.16320972982794046, "reward_total_mean": 0.5171016454696655, "reward_meter_mean": 0.6335141658782959, "reward_meter_std": 0.4142892062664032, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9450810551643372, "reward_repeat_soft_std": 0.02819715067744255, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.5171016454696655, "reward_total_composite_std": 0.114821657538414} {"timestamp_utc": "2026-04-13T11:03:46Z", "mode": "train", "global_step": 1511, "epoch": 0.15178302360622803, "loss": -0.1045, "grad_norm": 2.0721373558044434, "learning_rate": 5.424242424242425e-06, "num_tokens": 2670635.0, "completions/mean_length": 100.5, "completions/min_length": 37.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 41.71428680419922, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.7495286464691162, "rewards/meter/std": 0.22441412508487701, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9604498147964478, "rewards/repeat_soft/std": 0.04676219820976257, "rewards/judge_quality/mean": 0.33375000953674316, "rewards/judge_quality/std": 0.16159361600875854, "rewards/total_composite/mean": 0.47083914279937744, "rewards/total_composite/std": 0.201918825507164, "reward": 0.47083914279937744, "reward_std": 0.2019188106060028, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13865235447883606, "sampling/sampling_logp_difference/max": 1.438107967376709, "sampling/importance_sampling_ratio/min": 0.23737645149230957, "sampling/importance_sampling_ratio/mean": 1.026513934135437, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7795788422226906, "clip_ratio/low_mean": 0.04449760727584362, "clip_ratio/low_min": 0.04449760727584362, "clip_ratio/high_mean": 0.0813118212390691, "clip_ratio/high_max": 0.0813118212390691, "clip_ratio/region_mean": 0.12580942851491272, "reward_total_mean": 0.47083914279937744, "reward_meter_mean": 0.7495286464691162, "reward_meter_std": 0.22441412508487701, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9604498147964478, "reward_repeat_soft_std": 0.04676219820976257, "reward_judge_quality_mean": 0.33375000953674316, "reward_judge_quality_std": 0.16159361600875854, "reward_total_composite_mean": 0.47083914279937744, "reward_total_composite_std": 0.201918825507164} {"timestamp_utc": "2026-04-13T11:03:52Z", "mode": "train", "global_step": 1512, "epoch": 0.1518834756403817, "loss": 0.0172, "grad_norm": 13.740234375, "learning_rate": 5.421212121212122e-06, "num_tokens": 2672290.0, "completions/mean_length": 41.875, "completions/min_length": 36.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.875, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.449136346578598, "rewards/meter/std": 0.31989529728889465, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9736288785934448, "rewards/repeat_soft/std": 0.027965646237134933, "rewards/judge_quality/mean": 0.6050000190734863, "rewards/judge_quality/std": 0.22025960683822632, "rewards/total_composite/mean": 0.5165993571281433, "rewards/total_composite/std": 0.12132763862609863, "reward": 0.5165993571281433, "reward_std": 0.12132763862609863, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14902806282043457, "sampling/sampling_logp_difference/max": 1.2676098346710205, "sampling/importance_sampling_ratio/min": 0.28150367736816406, "sampling/importance_sampling_ratio/mean": 1.0072979927062988, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8480989933013916, "clip_ratio/low_mean": 0.059883545152843, "clip_ratio/low_min": 0.059883545152843, "clip_ratio/high_mean": 0.07934731431305408, "clip_ratio/high_max": 0.07934731431305408, "clip_ratio/region_mean": 0.13923085946589708, "reward_total_mean": 0.5165993571281433, "reward_meter_mean": 0.449136346578598, "reward_meter_std": 0.31989529728889465, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9736288785934448, "reward_repeat_soft_std": 0.027965646237134933, "reward_judge_quality_mean": 0.6050000190734863, "reward_judge_quality_std": 0.22025960683822632, "reward_total_composite_mean": 0.5165993571281433, "reward_total_composite_std": 0.12132763862609863} {"timestamp_utc": "2026-04-13T11:04:03Z", "mode": "train", "global_step": 1513, "epoch": 0.15198392767453542, "loss": -0.189, "grad_norm": 1.9482091665267944, "learning_rate": 5.418181818181819e-06, "num_tokens": 2674183.0, "completions/mean_length": 207.625, "completions/min_length": 90.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 106.16667175292969, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.6227824687957764, "rewards/meter/std": 0.3599480390548706, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.18600596487522125, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9174457788467407, "rewards/repeat_soft/std": 0.03720393404364586, "rewards/judge_quality/mean": 0.32749998569488525, "rewards/judge_quality/std": 0.17127670347690582, "rewards/total_composite/mean": 0.4073261320590973, "rewards/total_composite/std": 0.2590354084968567, "reward": 0.4073261320590973, "reward_std": 0.2590354084968567, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12539781630039215, "sampling/sampling_logp_difference/max": 2.0272881984710693, "sampling/importance_sampling_ratio/min": 0.13169217109680176, "sampling/importance_sampling_ratio/mean": 1.0209314823150635, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4624735414981842, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09595593344420195, "clip_ratio/high_max": 0.09595593344420195, "clip_ratio/region_mean": 0.09595593344420195, "reward_total_mean": 0.4073261320590973, "reward_meter_mean": 0.6227824687957764, "reward_meter_std": 0.3599480390548706, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.18600596487522125, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9174457788467407, "reward_repeat_soft_std": 0.03720393404364586, "reward_judge_quality_mean": 0.32749998569488525, "reward_judge_quality_std": 0.17127670347690582, "reward_total_composite_mean": 0.4073261320590973, "reward_total_composite_std": 0.2590354084968567} {"timestamp_utc": "2026-04-13T11:04:10Z", "mode": "train", "global_step": 1514, "epoch": 0.1520843797086891, "loss": 0.0397, "grad_norm": 14.658851623535156, "learning_rate": 5.415151515151515e-06, "num_tokens": 2675842.0, "completions/mean_length": 42.375, "completions/min_length": 34.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.375, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.40207529067993164, "rewards/meter/std": 0.38992664217948914, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9891808032989502, "rewards/repeat_soft/std": 0.009330120868980885, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.4599241614341736, "rewards/total_composite/std": 0.10620740801095963, "reward": 0.4599241614341736, "reward_std": 0.10620740801095963, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15452642738819122, "sampling/sampling_logp_difference/max": 1.1953868865966797, "sampling/importance_sampling_ratio/min": 0.3025868833065033, "sampling/importance_sampling_ratio/mean": 1.004881739616394, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.845390260219574, "clip_ratio/low_mean": 0.11816397495567799, "clip_ratio/low_min": 0.11816397495567799, "clip_ratio/high_mean": 0.0432773120701313, "clip_ratio/high_max": 0.0432773120701313, "clip_ratio/region_mean": 0.1614412870258093, "reward_total_mean": 0.4599241614341736, "reward_meter_mean": 0.40207529067993164, "reward_meter_std": 0.38992664217948914, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9891808032989502, "reward_repeat_soft_std": 0.009330120868980885, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.4599241614341736, "reward_total_composite_std": 0.10620740801095963} {"timestamp_utc": "2026-04-13T11:04:16Z", "mode": "train", "global_step": 1515, "epoch": 0.15218483174284278, "loss": -0.0282, "grad_norm": 10.259291648864746, "learning_rate": 5.412121212121213e-06, "num_tokens": 2677438.0, "completions/mean_length": 32.5, "completions/min_length": 29.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.5, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.7878463864326477, "rewards/meter/std": 0.2908877432346344, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9157747030258179, "rewards/repeat_soft/std": 0.08889061957597733, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5524482727050781, "rewards/total_composite/std": 0.07280430197715759, "reward": 0.5524482727050781, "reward_std": 0.07280430942773819, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08970874547958374, "sampling/sampling_logp_difference/max": 1.055037260055542, "sampling/importance_sampling_ratio/min": 0.348179429769516, "sampling/importance_sampling_ratio/mean": 0.9857891201972961, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.34806801564991474, "clip_ratio/low_mean": 0.01293103490024805, "clip_ratio/low_min": 0.01293103490024805, "clip_ratio/high_mean": 0.07328375522047281, "clip_ratio/high_max": 0.07328375522047281, "clip_ratio/region_mean": 0.08621479012072086, "reward_total_mean": 0.5524482727050781, "reward_meter_mean": 0.7878463864326477, "reward_meter_std": 0.2908877432346344, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9157747030258179, "reward_repeat_soft_std": 0.08889061957597733, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5524482727050781, "reward_total_composite_std": 0.07280430197715759} {"timestamp_utc": "2026-04-13T11:04:21Z", "mode": "train", "global_step": 1516, "epoch": 0.1522852837769965, "loss": 0.0945, "grad_norm": 10.050907135009766, "learning_rate": 5.409090909090909e-06, "num_tokens": 2678942.0, "completions/mean_length": 45.0, "completions/min_length": 37.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.0, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.9644712805747986, "rewards/meter/std": 0.04411641135811806, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8823840618133545, "rewards/repeat_soft/std": 0.07079489529132843, "rewards/judge_quality/mean": 0.5900000333786011, "rewards/judge_quality/std": 0.27994900941848755, "rewards/total_composite/mean": 0.701180100440979, "rewards/total_composite/std": 0.1814952939748764, "reward": 0.701180100440979, "reward_std": 0.1814952939748764, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10215334594249725, "sampling/sampling_logp_difference/max": 2.1007111072540283, "sampling/importance_sampling_ratio/min": 0.12236937880516052, "sampling/importance_sampling_ratio/mean": 1.007930874824524, "sampling/importance_sampling_ratio/max": 1.9660403728485107, "entropy": 0.5422038696706295, "clip_ratio/low_mean": 0.04473423445597291, "clip_ratio/low_min": 0.04473423445597291, "clip_ratio/high_mean": 0.04288395494222641, "clip_ratio/high_max": 0.04288395494222641, "clip_ratio/region_mean": 0.08761818939819932, "reward_total_mean": 0.701180100440979, "reward_meter_mean": 0.9644712805747986, "reward_meter_std": 0.04411641135811806, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8823840618133545, "reward_repeat_soft_std": 0.07079489529132843, "reward_judge_quality_mean": 0.5900000333786011, "reward_judge_quality_std": 0.27994900941848755, "reward_total_composite_mean": 0.701180100440979, "reward_total_composite_std": 0.1814952939748764} {"timestamp_utc": "2026-04-13T11:04:27Z", "mode": "train", "global_step": 1517, "epoch": 0.15238573581115017, "loss": 0.1147, "grad_norm": 20.944910049438477, "learning_rate": 5.406060606060607e-06, "num_tokens": 2680412.0, "completions/mean_length": 23.75, "completions/min_length": 20.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.75, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.7645701169967651, "rewards/meter/std": 0.3575473725795746, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9476451277732849, "rewards/repeat_soft/std": 0.021861007437109947, "rewards/judge_quality/mean": 0.5975000262260437, "rewards/judge_quality/std": 0.2750454545021057, "rewards/total_composite/mean": 0.6685272455215454, "rewards/total_composite/std": 0.2298787236213684, "reward": 0.6685272455215454, "reward_std": 0.2298787236213684, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1604284942150116, "sampling/sampling_logp_difference/max": 1.1559343338012695, "sampling/importance_sampling_ratio/min": 0.31476327776908875, "sampling/importance_sampling_ratio/mean": 1.0079511404037476, "sampling/importance_sampling_ratio/max": 1.9789609909057617, "entropy": 1.0180640891194344, "clip_ratio/low_mean": 0.09042960871011019, "clip_ratio/low_min": 0.09042960871011019, "clip_ratio/high_mean": 0.038977273274213076, "clip_ratio/high_max": 0.038977273274213076, "clip_ratio/region_mean": 0.12940688198432326, "reward_total_mean": 0.6685272455215454, "reward_meter_mean": 0.7645701169967651, "reward_meter_std": 0.3575473725795746, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9476451277732849, "reward_repeat_soft_std": 0.021861007437109947, "reward_judge_quality_mean": 0.5975000262260437, "reward_judge_quality_std": 0.2750454545021057, "reward_total_composite_mean": 0.6685272455215454, "reward_total_composite_std": 0.2298787236213684} {"timestamp_utc": "2026-04-13T11:04:34Z", "mode": "train", "global_step": 1518, "epoch": 0.15248618784530388, "loss": -0.0043, "grad_norm": 7.887856960296631, "learning_rate": 5.4030303030303036e-06, "num_tokens": 2682797.0, "completions/mean_length": 106.125, "completions/min_length": 78.0, "completions/max_length": 126.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.125, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 126.0, "rewards/meter/mean": 0.9521242380142212, "rewards/meter/std": 0.03605160117149353, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.928634762763977, "rewards/repeat_soft/std": 0.04874228313565254, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.6227474212646484, "rewards/total_composite/std": 0.0639571100473404, "reward": 0.6227474212646484, "reward_std": 0.0639571025967598, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13802793622016907, "sampling/sampling_logp_difference/max": 2.396497964859009, "sampling/importance_sampling_ratio/min": 0.09103620797395706, "sampling/importance_sampling_ratio/mean": 1.0033305883407593, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6701874136924744, "clip_ratio/low_mean": 0.10489683970808983, "clip_ratio/low_min": 0.10489683970808983, "clip_ratio/high_mean": 0.013392857275903225, "clip_ratio/high_max": 0.013392857275903225, "clip_ratio/region_mean": 0.11828969698399305, "reward_total_mean": 0.6227474212646484, "reward_meter_mean": 0.9521242380142212, "reward_meter_std": 0.03605160117149353, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.928634762763977, "reward_repeat_soft_std": 0.04874228313565254, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.6227474212646484, "reward_total_composite_std": 0.0639571100473404} {"timestamp_utc": "2026-04-13T11:04:41Z", "mode": "train", "global_step": 1519, "epoch": 0.15258663987945756, "loss": 0.1012, "grad_norm": 10.09461498260498, "learning_rate": 5.400000000000001e-06, "num_tokens": 2685120.0, "completions/mean_length": 106.375, "completions/min_length": 85.0, "completions/max_length": 125.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.375, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.7603713870048523, "rewards/meter/std": 0.3669488728046417, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9532291889190674, "rewards/repeat_soft/std": 0.042330704629421234, "rewards/judge_quality/mean": 0.7325000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.6824182271957397, "rewards/total_composite/std": 0.20744413137435913, "reward": 0.6824182271957397, "reward_std": 0.20744411647319794, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15000297129154205, "sampling/sampling_logp_difference/max": 2.788607120513916, "sampling/importance_sampling_ratio/min": 0.13063034415245056, "sampling/importance_sampling_ratio/mean": 1.01371431350708, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9093747064471245, "clip_ratio/low_mean": 0.08507308643311262, "clip_ratio/low_min": 0.08507308643311262, "clip_ratio/high_mean": 0.049013751558959484, "clip_ratio/high_max": 0.049013751558959484, "clip_ratio/region_mean": 0.1340868379920721, "reward_total_mean": 0.6824182271957397, "reward_meter_mean": 0.7603713870048523, "reward_meter_std": 0.3669488728046417, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9532291889190674, "reward_repeat_soft_std": 0.042330704629421234, "reward_judge_quality_mean": 0.7325000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.6824182271957397, "reward_total_composite_std": 0.20744413137435913} {"timestamp_utc": "2026-04-13T11:04:48Z", "mode": "train", "global_step": 1520, "epoch": 0.15268709191361124, "loss": -0.0249, "grad_norm": 7.378472328186035, "learning_rate": 5.396969696969697e-06, "num_tokens": 2687601.0, "completions/mean_length": 121.125, "completions/min_length": 93.0, "completions/max_length": 135.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 121.125, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 135.0, "rewards/meter/mean": 0.926666259765625, "rewards/meter/std": 0.16308124363422394, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9598362445831299, "rewards/repeat_soft/std": 0.024580257013440132, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.6049239039421082, "rewards/total_composite/std": 0.10893219709396362, "reward": 0.6049239039421082, "reward_std": 0.10893220454454422, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12834319472312927, "sampling/sampling_logp_difference/max": 1.499781608581543, "sampling/importance_sampling_ratio/min": 0.2231789082288742, "sampling/importance_sampling_ratio/mean": 1.0122907161712646, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7687502652406693, "clip_ratio/low_mean": 0.09747287072241306, "clip_ratio/low_min": 0.09747287072241306, "clip_ratio/high_mean": 0.031536294147372246, "clip_ratio/high_max": 0.031536294147372246, "clip_ratio/region_mean": 0.1290091648697853, "reward_total_mean": 0.6049239039421082, "reward_meter_mean": 0.926666259765625, "reward_meter_std": 0.16308124363422394, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9598362445831299, "reward_repeat_soft_std": 0.024580257013440132, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.6049239039421082, "reward_total_composite_std": 0.10893219709396362} {"timestamp_utc": "2026-04-13T11:04:54Z", "mode": "train", "global_step": 1521, "epoch": 0.15278754394776495, "loss": -0.0609, "grad_norm": 11.856502532958984, "learning_rate": 5.3939393939393945e-06, "num_tokens": 2689306.0, "completions/mean_length": 53.125, "completions/min_length": 41.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.125, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9026252627372742, "rewards/meter/std": 0.233467236161232, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9118351936340332, "rewards/repeat_soft/std": 0.05899510160088539, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.10507651418447495, "rewards/total_composite/mean": 0.6094926595687866, "rewards/total_composite/std": 0.09445804357528687, "reward": 0.6094926595687866, "reward_std": 0.09445804357528687, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11438848823308945, "sampling/sampling_logp_difference/max": 2.0489282608032227, "sampling/importance_sampling_ratio/min": 0.12887294590473175, "sampling/importance_sampling_ratio/mean": 1.0063875913619995, "sampling/importance_sampling_ratio/max": 1.8470425605773926, "entropy": 0.7874886989593506, "clip_ratio/low_mean": 0.05083679733797908, "clip_ratio/low_min": 0.05083679733797908, "clip_ratio/high_mean": 0.04340798268094659, "clip_ratio/high_max": 0.04340798268094659, "clip_ratio/region_mean": 0.09424478001892567, "reward_total_mean": 0.6094926595687866, "reward_meter_mean": 0.9026252627372742, "reward_meter_std": 0.233467236161232, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9118351936340332, "reward_repeat_soft_std": 0.05899510160088539, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.10507651418447495, "reward_total_composite_mean": 0.6094926595687866, "reward_total_composite_std": 0.09445804357528687} {"timestamp_utc": "2026-04-13T11:05:00Z", "mode": "train", "global_step": 1522, "epoch": 0.15288799598191863, "loss": 0.0047, "grad_norm": 20.712505340576172, "learning_rate": 5.390909090909091e-06, "num_tokens": 2690716.0, "completions/mean_length": 26.25, "completions/min_length": 21.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.25, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.565726637840271, "rewards/meter/std": 0.45630767941474915, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9610613584518433, "rewards/repeat_soft/std": 0.004068940877914429, "rewards/judge_quality/mean": 0.8237500190734863, "rewards/judge_quality/std": 0.2722361385822296, "rewards/total_composite/mean": 0.6389440894126892, "rewards/total_composite/std": 0.3503834903240204, "reward": 0.6389440894126892, "reward_std": 0.350383460521698, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17062167823314667, "sampling/sampling_logp_difference/max": 1.8133325576782227, "sampling/importance_sampling_ratio/min": 0.1631096601486206, "sampling/importance_sampling_ratio/mean": 1.0215020179748535, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5077793076634407, "clip_ratio/low_mean": 0.029363354202359915, "clip_ratio/low_min": 0.029363354202359915, "clip_ratio/high_mean": 0.08101852051913738, "clip_ratio/high_max": 0.08101852051913738, "clip_ratio/region_mean": 0.1103818747214973, "reward_total_mean": 0.6389440894126892, "reward_meter_mean": 0.565726637840271, "reward_meter_std": 0.45630767941474915, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9610613584518433, "reward_repeat_soft_std": 0.004068940877914429, "reward_judge_quality_mean": 0.8237500190734863, "reward_judge_quality_std": 0.2722361385822296, "reward_total_composite_mean": 0.6389440894126892, "reward_total_composite_std": 0.3503834903240204} {"timestamp_utc": "2026-04-13T11:05:06Z", "mode": "train", "global_step": 1523, "epoch": 0.15298844801607234, "loss": 0.0921, "grad_norm": 11.37076473236084, "learning_rate": 5.387878787878789e-06, "num_tokens": 2692426.0, "completions/mean_length": 50.75, "completions/min_length": 36.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.75, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.7785061597824097, "rewards/meter/std": 0.3467750549316406, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9427660703659058, "rewards/repeat_soft/std": 0.044391099363565445, "rewards/judge_quality/mean": 0.6112500429153442, "rewards/judge_quality/std": 0.25587037205696106, "rewards/total_composite/mean": 0.648284912109375, "rewards/total_composite/std": 0.1843511015176773, "reward": 0.648284912109375, "reward_std": 0.1843511164188385, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14299741387367249, "sampling/sampling_logp_difference/max": 3.153827428817749, "sampling/importance_sampling_ratio/min": 0.04268842563033104, "sampling/importance_sampling_ratio/mean": 0.9996667504310608, "sampling/importance_sampling_ratio/max": 1.9937615394592285, "entropy": 0.9337623193860054, "clip_ratio/low_mean": 0.07621855894103646, "clip_ratio/low_min": 0.07621855894103646, "clip_ratio/high_mean": 0.031250000931322575, "clip_ratio/high_max": 0.031250000931322575, "clip_ratio/region_mean": 0.10746855987235904, "reward_total_mean": 0.648284912109375, "reward_meter_mean": 0.7785061597824097, "reward_meter_std": 0.3467750549316406, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9427660703659058, "reward_repeat_soft_std": 0.044391099363565445, "reward_judge_quality_mean": 0.6112500429153442, "reward_judge_quality_std": 0.25587037205696106, "reward_total_composite_mean": 0.648284912109375, "reward_total_composite_std": 0.1843511015176773} {"timestamp_utc": "2026-04-13T11:05:14Z", "mode": "train", "global_step": 1524, "epoch": 0.15308890005022602, "loss": 0.1018, "grad_norm": 7.0124335289001465, "learning_rate": 5.384848484848485e-06, "num_tokens": 2694722.0, "completions/mean_length": 111.0, "completions/min_length": 88.0, "completions/max_length": 134.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 111.0, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 134.0, "rewards/meter/mean": 0.7280535101890564, "rewards/meter/std": 0.3529597818851471, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7642220854759216, "rewards/repeat_soft/std": 0.08997632563114166, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.49593979120254517, "rewards/total_composite/std": 0.09022112190723419, "reward": 0.49593979120254517, "reward_std": 0.090221107006073, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08519119769334793, "sampling/sampling_logp_difference/max": 1.982034683227539, "sampling/importance_sampling_ratio/min": 0.1377885937690735, "sampling/importance_sampling_ratio/mean": 0.9952489733695984, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4152562692761421, "clip_ratio/low_mean": 0.03927290253341198, "clip_ratio/low_min": 0.03927290253341198, "clip_ratio/high_mean": 0.045991506427526474, "clip_ratio/high_max": 0.045991506427526474, "clip_ratio/region_mean": 0.08526440896093845, "reward_total_mean": 0.49593979120254517, "reward_meter_mean": 0.7280535101890564, "reward_meter_std": 0.3529597818851471, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7642220854759216, "reward_repeat_soft_std": 0.08997632563114166, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.49593979120254517, "reward_total_composite_std": 0.09022112190723419} {"timestamp_utc": "2026-04-13T11:05:20Z", "mode": "train", "global_step": 1525, "epoch": 0.1531893520843797, "loss": 0.0811, "grad_norm": 12.178094863891602, "learning_rate": 5.381818181818183e-06, "num_tokens": 2696332.0, "completions/mean_length": 51.25, "completions/min_length": 44.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.25, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.8399039506912231, "rewards/meter/std": 0.21824374794960022, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9909389019012451, "rewards/repeat_soft/std": 0.00940681528300047, "rewards/judge_quality/mean": 0.7437500357627869, "rewards/judge_quality/std": 0.2083909660577774, "rewards/total_composite/mean": 0.752814531326294, "rewards/total_composite/std": 0.15112841129302979, "reward": 0.752814531326294, "reward_std": 0.15112842619419098, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11794614046812057, "sampling/sampling_logp_difference/max": 1.3004674911499023, "sampling/importance_sampling_ratio/min": 0.27240443229675293, "sampling/importance_sampling_ratio/mean": 1.02642023563385, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7635732442140579, "clip_ratio/low_mean": 0.06484845839440823, "clip_ratio/low_min": 0.06484845839440823, "clip_ratio/high_mean": 0.07267182040959597, "clip_ratio/high_max": 0.07267182040959597, "clip_ratio/region_mean": 0.1375202788040042, "reward_total_mean": 0.752814531326294, "reward_meter_mean": 0.8399039506912231, "reward_meter_std": 0.21824374794960022, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9909389019012451, "reward_repeat_soft_std": 0.00940681528300047, "reward_judge_quality_mean": 0.7437500357627869, "reward_judge_quality_std": 0.2083909660577774, "reward_total_composite_mean": 0.752814531326294, "reward_total_composite_std": 0.15112841129302979} {"timestamp_utc": "2026-04-13T11:05:26Z", "mode": "train", "global_step": 1526, "epoch": 0.1532898041185334, "loss": 0.0942, "grad_norm": 9.61211109161377, "learning_rate": 5.378787878787879e-06, "num_tokens": 2698420.0, "completions/mean_length": 84.0, "completions/min_length": 68.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.0, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.6344290375709534, "rewards/meter/std": 0.37753942608833313, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9780125021934509, "rewards/repeat_soft/std": 0.034678518772125244, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.49886345863342285, "rewards/total_composite/std": 0.09963525831699371, "reward": 0.49886345863342285, "reward_std": 0.09963525831699371, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16740047931671143, "sampling/sampling_logp_difference/max": 3.027803897857666, "sampling/importance_sampling_ratio/min": 0.04842185974121094, "sampling/importance_sampling_ratio/mean": 1.015329122543335, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4986537620425224, "clip_ratio/low_mean": 0.07367149833589792, "clip_ratio/low_min": 0.07367149833589792, "clip_ratio/high_mean": 0.07234195433557034, "clip_ratio/high_max": 0.07234195433557034, "clip_ratio/region_mean": 0.14601345267146826, "reward_total_mean": 0.49886345863342285, "reward_meter_mean": 0.6344290375709534, "reward_meter_std": 0.37753942608833313, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9780125021934509, "reward_repeat_soft_std": 0.034678518772125244, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.49886345863342285, "reward_total_composite_std": 0.09963525831699371} {"timestamp_utc": "2026-04-13T11:05:33Z", "mode": "train", "global_step": 1527, "epoch": 0.1533902561526871, "loss": 0.0509, "grad_norm": 9.31325626373291, "learning_rate": 5.375757575757576e-06, "num_tokens": 2700394.0, "completions/mean_length": 83.75, "completions/min_length": 64.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 83.75, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.8043165802955627, "rewards/meter/std": 0.2480888068675995, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8991732001304626, "rewards/repeat_soft/std": 0.083254374563694, "rewards/judge_quality/mean": 0.5900000333786011, "rewards/judge_quality/std": 0.1891333907842636, "rewards/total_composite/mean": 0.6492887139320374, "rewards/total_composite/std": 0.14161844551563263, "reward": 0.6492887139320374, "reward_std": 0.14161844551563263, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11522864550352097, "sampling/sampling_logp_difference/max": 1.8459806442260742, "sampling/importance_sampling_ratio/min": 0.15787042677402496, "sampling/importance_sampling_ratio/mean": 1.0108704566955566, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.574681032449007, "clip_ratio/low_mean": 0.06570903304964304, "clip_ratio/low_min": 0.06570903304964304, "clip_ratio/high_mean": 0.05417239014059305, "clip_ratio/high_max": 0.05417239014059305, "clip_ratio/region_mean": 0.11988142319023609, "reward_total_mean": 0.6492887139320374, "reward_meter_mean": 0.8043165802955627, "reward_meter_std": 0.2480888068675995, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8991732001304626, "reward_repeat_soft_std": 0.083254374563694, "reward_judge_quality_mean": 0.5900000333786011, "reward_judge_quality_std": 0.1891333907842636, "reward_total_composite_mean": 0.6492887139320374, "reward_total_composite_std": 0.14161844551563263} {"timestamp_utc": "2026-04-13T11:05:40Z", "mode": "train", "global_step": 1528, "epoch": 0.1534907081868408, "loss": 0.0588, "grad_norm": 7.919285774230957, "learning_rate": 5.372727272727273e-06, "num_tokens": 2702601.0, "completions/mean_length": 104.875, "completions/min_length": 85.0, "completions/max_length": 137.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 104.875, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 137.0, "rewards/meter/mean": 0.9438287019729614, "rewards/meter/std": 0.11375102400779724, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8729192018508911, "rewards/repeat_soft/std": 0.12133605778217316, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.5657995939254761, "rewards/total_composite/std": 0.06492535769939423, "reward": 0.5657995939254761, "reward_std": 0.06492535769939423, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11691422015428543, "sampling/sampling_logp_difference/max": 1.281205177307129, "sampling/importance_sampling_ratio/min": 0.2777024209499359, "sampling/importance_sampling_ratio/mean": 1.0154949426651, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8299001902341843, "clip_ratio/low_mean": 0.04468996450304985, "clip_ratio/low_min": 0.04468996450304985, "clip_ratio/high_mean": 0.0835497984662652, "clip_ratio/high_max": 0.0835497984662652, "clip_ratio/region_mean": 0.12823976296931505, "reward_total_mean": 0.5657995939254761, "reward_meter_mean": 0.9438287019729614, "reward_meter_std": 0.11375102400779724, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8729192018508911, "reward_repeat_soft_std": 0.12133605778217316, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.5657995939254761, "reward_total_composite_std": 0.06492535769939423} {"timestamp_utc": "2026-04-13T11:05:48Z", "mode": "train", "global_step": 1529, "epoch": 0.15359116022099448, "loss": 0.0283, "grad_norm": 6.661654949188232, "learning_rate": 5.36969696969697e-06, "num_tokens": 2704969.0, "completions/mean_length": 122.0, "completions/min_length": 111.0, "completions/max_length": 136.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.0, "completions/min_terminated_length": 111.0, "completions/max_terminated_length": 136.0, "rewards/meter/mean": 0.562111496925354, "rewards/meter/std": 0.3219490349292755, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8385282754898071, "rewards/repeat_soft/std": 0.077992282807827, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.45416495203971863, "rewards/total_composite/std": 0.08038046211004257, "reward": 0.45416495203971863, "reward_std": 0.08038046956062317, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09357380121946335, "sampling/sampling_logp_difference/max": 2.4090735912323, "sampling/importance_sampling_ratio/min": 0.08989854902029037, "sampling/importance_sampling_ratio/mean": 1.0042166709899902, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5388602800667286, "clip_ratio/low_mean": 0.026751196943223476, "clip_ratio/low_min": 0.026751196943223476, "clip_ratio/high_mean": 0.058976574800908566, "clip_ratio/high_max": 0.058976574800908566, "clip_ratio/region_mean": 0.08572777174413204, "reward_total_mean": 0.45416495203971863, "reward_meter_mean": 0.562111496925354, "reward_meter_std": 0.3219490349292755, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8385282754898071, "reward_repeat_soft_std": 0.077992282807827, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.45416495203971863, "reward_total_composite_std": 0.08038046211004257} {"timestamp_utc": "2026-04-13T11:05:59Z", "mode": "train", "global_step": 1530, "epoch": 0.15369161225514816, "loss": -0.0947, "grad_norm": 3.3263089656829834, "learning_rate": 5.366666666666666e-06, "num_tokens": 2706594.0, "completions/mean_length": 105.125, "completions/min_length": 40.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 47.000003814697266, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.5275822281837463, "rewards/meter/std": 0.42082855105400085, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9427675008773804, "rewards/repeat_soft/std": 0.09034770727157593, "rewards/judge_quality/mean": 0.6312500238418579, "rewards/judge_quality/std": 0.3341701030731201, "rewards/total_composite/mean": 0.5328185558319092, "rewards/total_composite/std": 0.2762130796909332, "reward": 0.5328185558319092, "reward_std": 0.27621304988861084, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11439567804336548, "sampling/sampling_logp_difference/max": 1.2095632553100586, "sampling/importance_sampling_ratio/min": 0.2983275353908539, "sampling/importance_sampling_ratio/mean": 1.0066003799438477, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.531870748847723, "clip_ratio/low_mean": 0.04078907240182161, "clip_ratio/low_min": 0.04078907240182161, "clip_ratio/high_mean": 0.0648478849325329, "clip_ratio/high_max": 0.0648478849325329, "clip_ratio/region_mean": 0.10563695733435452, "reward_total_mean": 0.5328185558319092, "reward_meter_mean": 0.5275822281837463, "reward_meter_std": 0.42082855105400085, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9427675008773804, "reward_repeat_soft_std": 0.09034770727157593, "reward_judge_quality_mean": 0.6312500238418579, "reward_judge_quality_std": 0.3341701030731201, "reward_total_composite_mean": 0.5328185558319092, "reward_total_composite_std": 0.2762130796909332} {"timestamp_utc": "2026-04-13T11:06:06Z", "mode": "train", "global_step": 1531, "epoch": 0.15379206428930187, "loss": 0.0017, "grad_norm": 13.402848243713379, "learning_rate": 5.3636363636363645e-06, "num_tokens": 2708116.0, "completions/mean_length": 40.25, "completions/min_length": 36.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.25, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.5361714363098145, "rewards/meter/std": 0.3404063284397125, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9781786799430847, "rewards/repeat_soft/std": 0.028630316257476807, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.4972688853740692, "rewards/total_composite/std": 0.09261411428451538, "reward": 0.4972688853740692, "reward_std": 0.09261411428451538, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11025577038526535, "sampling/sampling_logp_difference/max": 2.0586910247802734, "sampling/importance_sampling_ratio/min": 0.1276209056377411, "sampling/importance_sampling_ratio/mean": 1.0056194067001343, "sampling/importance_sampling_ratio/max": 1.8145413398742676, "entropy": 0.7311753779649734, "clip_ratio/low_mean": 0.035480367951095104, "clip_ratio/low_min": 0.035480367951095104, "clip_ratio/high_mean": 0.0829462818801403, "clip_ratio/high_max": 0.0829462818801403, "clip_ratio/region_mean": 0.11842664983123541, "reward_total_mean": 0.4972688853740692, "reward_meter_mean": 0.5361714363098145, "reward_meter_std": 0.3404063284397125, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9781786799430847, "reward_repeat_soft_std": 0.028630316257476807, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.4972688853740692, "reward_total_composite_std": 0.09261411428451538} {"timestamp_utc": "2026-04-13T11:06:12Z", "mode": "train", "global_step": 1532, "epoch": 0.15389251632345555, "loss": -0.0447, "grad_norm": 14.792496681213379, "learning_rate": 5.360606060606061e-06, "num_tokens": 2709704.0, "completions/mean_length": 37.5, "completions/min_length": 32.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.5, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.9494877457618713, "rewards/meter/std": 0.009832732379436493, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9202165007591248, "rewards/repeat_soft/std": 0.09014234691858292, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.6734165549278259, "rewards/total_composite/std": 0.146859809756279, "reward": 0.6734165549278259, "reward_std": 0.146859809756279, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10582664608955383, "sampling/sampling_logp_difference/max": 1.6556751728057861, "sampling/importance_sampling_ratio/min": 0.19096307456493378, "sampling/importance_sampling_ratio/mean": 1.0053645372390747, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5843740478157997, "clip_ratio/low_mean": 0.0696247685700655, "clip_ratio/low_min": 0.0696247685700655, "clip_ratio/high_mean": 0.025600685738027096, "clip_ratio/high_max": 0.025600685738027096, "clip_ratio/region_mean": 0.0952254543080926, "reward_total_mean": 0.6734165549278259, "reward_meter_mean": 0.9494877457618713, "reward_meter_std": 0.009832732379436493, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9202165007591248, "reward_repeat_soft_std": 0.09014234691858292, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.6734165549278259, "reward_total_composite_std": 0.146859809756279} {"timestamp_utc": "2026-04-13T11:06:18Z", "mode": "train", "global_step": 1533, "epoch": 0.15399296835760923, "loss": 0.0066, "grad_norm": 11.148467063903809, "learning_rate": 5.357575757575758e-06, "num_tokens": 2711471.0, "completions/mean_length": 46.875, "completions/min_length": 41.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.875, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9281042814254761, "rewards/meter/std": 0.14827781915664673, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9227937459945679, "rewards/repeat_soft/std": 0.053246546536684036, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5917915105819702, "rewards/total_composite/std": 0.03999544680118561, "reward": 0.5917915105819702, "reward_std": 0.0399954617023468, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14617781341075897, "sampling/sampling_logp_difference/max": 4.057716369628906, "sampling/importance_sampling_ratio/min": 0.0172884538769722, "sampling/importance_sampling_ratio/mean": 0.996834933757782, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8533198460936546, "clip_ratio/low_mean": 0.016735858283936977, "clip_ratio/low_min": 0.016735858283936977, "clip_ratio/high_mean": 0.07233824674040079, "clip_ratio/high_max": 0.07233824674040079, "clip_ratio/region_mean": 0.08907410502433777, "reward_total_mean": 0.5917915105819702, "reward_meter_mean": 0.9281042814254761, "reward_meter_std": 0.14827781915664673, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9227937459945679, "reward_repeat_soft_std": 0.053246546536684036, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5917915105819702, "reward_total_composite_std": 0.03999544680118561} {"timestamp_utc": "2026-04-13T11:06:25Z", "mode": "train", "global_step": 1534, "epoch": 0.15409342039176294, "loss": 0.0335, "grad_norm": 11.622137069702148, "learning_rate": 5.3545454545454546e-06, "num_tokens": 2713067.0, "completions/mean_length": 46.5, "completions/min_length": 36.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.5, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.7989070415496826, "rewards/meter/std": 0.30848050117492676, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9807281494140625, "rewards/repeat_soft/std": 0.02614726312458515, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.5687021613121033, "rewards/total_composite/std": 0.07431824505329132, "reward": 0.5687021613121033, "reward_std": 0.07431824505329132, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15608374774456024, "sampling/sampling_logp_difference/max": 1.595898151397705, "sampling/importance_sampling_ratio/min": 0.20841604471206665, "sampling/importance_sampling_ratio/mean": 0.9950181841850281, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.86784628033638, "clip_ratio/low_mean": 0.024980408139526844, "clip_ratio/low_min": 0.024980408139526844, "clip_ratio/high_mean": 0.11696935445070267, "clip_ratio/high_max": 0.11696935445070267, "clip_ratio/region_mean": 0.1419497625902295, "reward_total_mean": 0.5687021613121033, "reward_meter_mean": 0.7989070415496826, "reward_meter_std": 0.30848050117492676, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9807281494140625, "reward_repeat_soft_std": 0.02614726312458515, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.5687021613121033, "reward_total_composite_std": 0.07431824505329132} {"timestamp_utc": "2026-04-13T11:06:32Z", "mode": "train", "global_step": 1535, "epoch": 0.15419387242591662, "loss": 0.0702, "grad_norm": 13.182768821716309, "learning_rate": 5.351515151515152e-06, "num_tokens": 2714639.0, "completions/mean_length": 38.5, "completions/min_length": 28.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.5, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.7237865924835205, "rewards/meter/std": 0.2100200653076172, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9861932992935181, "rewards/repeat_soft/std": 0.038746047765016556, "rewards/judge_quality/mean": 0.6200000047683716, "rewards/judge_quality/std": 0.22677870094776154, "rewards/total_composite/mean": 0.6452229022979736, "rewards/total_composite/std": 0.15417109429836273, "reward": 0.6452229022979736, "reward_std": 0.15417106449604034, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14780864119529724, "sampling/sampling_logp_difference/max": 1.209925889968872, "sampling/importance_sampling_ratio/min": 0.29821938276290894, "sampling/importance_sampling_ratio/mean": 1.0319126844406128, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8952219486236572, "clip_ratio/low_mean": 0.07706950325518847, "clip_ratio/low_min": 0.07706950325518847, "clip_ratio/high_mean": 0.037288233172148466, "clip_ratio/high_max": 0.037288233172148466, "clip_ratio/region_mean": 0.11435773642733693, "reward_total_mean": 0.6452229022979736, "reward_meter_mean": 0.7237865924835205, "reward_meter_std": 0.2100200653076172, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9861932992935181, "reward_repeat_soft_std": 0.038746047765016556, "reward_judge_quality_mean": 0.6200000047683716, "reward_judge_quality_std": 0.22677870094776154, "reward_total_composite_mean": 0.6452229022979736, "reward_total_composite_std": 0.15417109429836273} {"timestamp_utc": "2026-04-13T11:06:38Z", "mode": "train", "global_step": 1536, "epoch": 0.15429432446007033, "loss": 0.0485, "grad_norm": 8.649129867553711, "learning_rate": 5.348484848484848e-06, "num_tokens": 2716736.0, "completions/mean_length": 90.125, "completions/min_length": 81.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.125, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.7510343790054321, "rewards/meter/std": 0.2199547290802002, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9506515264511108, "rewards/repeat_soft/std": 0.047333285212516785, "rewards/judge_quality/mean": 0.5149999856948853, "rewards/judge_quality/std": 0.1810288280248642, "rewards/total_composite/mean": 0.5881261825561523, "rewards/total_composite/std": 0.08761488646268845, "reward": 0.5881261825561523, "reward_std": 0.08761487901210785, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13279852271080017, "sampling/sampling_logp_difference/max": 2.762014865875244, "sampling/importance_sampling_ratio/min": 0.0631643682718277, "sampling/importance_sampling_ratio/mean": 1.008737325668335, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5726853087544441, "clip_ratio/low_mean": 0.043127247132360935, "clip_ratio/low_min": 0.043127247132360935, "clip_ratio/high_mean": 0.06832017423585057, "clip_ratio/high_max": 0.06832017423585057, "clip_ratio/region_mean": 0.11144742136821151, "reward_total_mean": 0.5881261825561523, "reward_meter_mean": 0.7510343790054321, "reward_meter_std": 0.2199547290802002, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9506515264511108, "reward_repeat_soft_std": 0.047333285212516785, "reward_judge_quality_mean": 0.5149999856948853, "reward_judge_quality_std": 0.1810288280248642, "reward_total_composite_mean": 0.5881261825561523, "reward_total_composite_std": 0.08761488646268845} {"timestamp_utc": "2026-04-13T11:06:46Z", "mode": "train", "global_step": 1537, "epoch": 0.154394776494224, "loss": 0.0057, "grad_norm": 6.371781826019287, "learning_rate": 5.3454545454545455e-06, "num_tokens": 2719387.0, "completions/mean_length": 142.375, "completions/min_length": 118.0, "completions/max_length": 161.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 142.375, "completions/min_terminated_length": 118.0, "completions/max_terminated_length": 161.0, "rewards/meter/mean": 0.7671759128570557, "rewards/meter/std": 0.2969895601272583, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8740166425704956, "rewards/repeat_soft/std": 0.04647092521190643, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.20860078930854797, "rewards/total_composite/mean": 0.5356192588806152, "rewards/total_composite/std": 0.16517037153244019, "reward": 0.5356192588806152, "reward_std": 0.16517037153244019, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12230098247528076, "sampling/sampling_logp_difference/max": 2.7857043743133545, "sampling/importance_sampling_ratio/min": 0.06168562173843384, "sampling/importance_sampling_ratio/mean": 1.016041874885559, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8719235211610794, "clip_ratio/low_mean": 0.037878320552408695, "clip_ratio/low_min": 0.037878320552408695, "clip_ratio/high_mean": 0.07058449182659388, "clip_ratio/high_max": 0.07058449182659388, "clip_ratio/region_mean": 0.10846281237900257, "reward_total_mean": 0.5356192588806152, "reward_meter_mean": 0.7671759128570557, "reward_meter_std": 0.2969895601272583, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8740166425704956, "reward_repeat_soft_std": 0.04647092521190643, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.20860078930854797, "reward_total_composite_mean": 0.5356192588806152, "reward_total_composite_std": 0.16517037153244019} {"timestamp_utc": "2026-04-13T11:06:52Z", "mode": "train", "global_step": 1538, "epoch": 0.1544952285283777, "loss": 0.036, "grad_norm": 6.371514797210693, "learning_rate": 5.342424242424244e-06, "num_tokens": 2721639.0, "completions/mean_length": 96.5, "completions/min_length": 81.0, "completions/max_length": 112.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.5, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 112.0, "rewards/meter/mean": 0.9475618600845337, "rewards/meter/std": 0.10585737228393555, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8539997935295105, "rewards/repeat_soft/std": 0.04982974752783775, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.6270068287849426, "rewards/total_composite/std": 0.1197318509221077, "reward": 0.6270068287849426, "reward_std": 0.1197318509221077, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10939937829971313, "sampling/sampling_logp_difference/max": 1.5694153308868408, "sampling/importance_sampling_ratio/min": 0.20816683769226074, "sampling/importance_sampling_ratio/mean": 1.0134103298187256, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6863407343626022, "clip_ratio/low_mean": 0.08676877291873097, "clip_ratio/low_min": 0.08676877291873097, "clip_ratio/high_mean": 0.010174418799579144, "clip_ratio/high_max": 0.010174418799579144, "clip_ratio/region_mean": 0.09694319171831012, "reward_total_mean": 0.6270068287849426, "reward_meter_mean": 0.9475618600845337, "reward_meter_std": 0.10585737228393555, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8539997935295105, "reward_repeat_soft_std": 0.04982974752783775, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.6270068287849426, "reward_total_composite_std": 0.1197318509221077} {"timestamp_utc": "2026-04-13T11:06:59Z", "mode": "train", "global_step": 1539, "epoch": 0.1545956805625314, "loss": 0.0437, "grad_norm": 14.528196334838867, "learning_rate": 5.33939393939394e-06, "num_tokens": 2723288.0, "completions/mean_length": 46.125, "completions/min_length": 41.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.125, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.8299557566642761, "rewards/meter/std": 0.3053044378757477, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9574869275093079, "rewards/repeat_soft/std": 0.04177146032452583, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.1527603268623352, "rewards/total_composite/mean": 0.5731433629989624, "rewards/total_composite/std": 0.12193405628204346, "reward": 0.5731433629989624, "reward_std": 0.12193406373262405, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1316116452217102, "sampling/sampling_logp_difference/max": 1.407602310180664, "sampling/importance_sampling_ratio/min": 0.24472936987876892, "sampling/importance_sampling_ratio/mean": 1.0131338834762573, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.930720329284668, "clip_ratio/low_mean": 0.04696767684072256, "clip_ratio/low_min": 0.04696767684072256, "clip_ratio/high_mean": 0.06994033884257078, "clip_ratio/high_max": 0.06994033884257078, "clip_ratio/region_mean": 0.11690801568329334, "reward_total_mean": 0.5731433629989624, "reward_meter_mean": 0.8299557566642761, "reward_meter_std": 0.3053044378757477, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9574869275093079, "reward_repeat_soft_std": 0.04177146032452583, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.1527603268623352, "reward_total_composite_mean": 0.5731433629989624, "reward_total_composite_std": 0.12193405628204346} {"timestamp_utc": "2026-04-13T11:07:07Z", "mode": "train", "global_step": 1540, "epoch": 0.15469613259668508, "loss": 0.0371, "grad_norm": 8.206191062927246, "learning_rate": 5.336363636363637e-06, "num_tokens": 2725521.0, "completions/mean_length": 110.125, "completions/min_length": 77.0, "completions/max_length": 149.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 110.125, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 149.0, "rewards/meter/mean": 0.8557904958724976, "rewards/meter/std": 0.24603213369846344, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8825615048408508, "rewards/repeat_soft/std": 0.1511286348104477, "rewards/judge_quality/mean": 0.4362499713897705, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.5596288442611694, "rewards/total_composite/std": 0.07174958288669586, "reward": 0.5596288442611694, "reward_std": 0.07174960523843765, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11753819137811661, "sampling/sampling_logp_difference/max": 2.8877174854278564, "sampling/importance_sampling_ratio/min": 0.055703211575746536, "sampling/importance_sampling_ratio/mean": 1.0019290447235107, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7809856422245502, "clip_ratio/low_mean": 0.017171660671010613, "clip_ratio/low_min": 0.017171660671010613, "clip_ratio/high_mean": 0.10712305922061205, "clip_ratio/high_max": 0.10712305922061205, "clip_ratio/region_mean": 0.12429471989162266, "reward_total_mean": 0.5596288442611694, "reward_meter_mean": 0.8557904958724976, "reward_meter_std": 0.24603213369846344, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8825615048408508, "reward_repeat_soft_std": 0.1511286348104477, "reward_judge_quality_mean": 0.4362499713897705, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.5596288442611694, "reward_total_composite_std": 0.07174958288669586} {"timestamp_utc": "2026-04-13T11:07:18Z", "mode": "train", "global_step": 1541, "epoch": 0.15479658463083878, "loss": -0.1553, "grad_norm": 3.9842867851257324, "learning_rate": 5.333333333333334e-06, "num_tokens": 2727819.0, "completions/mean_length": 163.25, "completions/min_length": 102.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 113.42857360839844, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 126.0, "rewards/meter/mean": 0.6968671679496765, "rewards/meter/std": 0.3204905390739441, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9399411678314209, "rewards/repeat_soft/std": 0.05086035653948784, "rewards/judge_quality/mean": 0.6687500476837158, "rewards/judge_quality/std": 0.29844537377357483, "rewards/total_composite/mean": 0.5980728268623352, "rewards/total_composite/std": 0.30698275566101074, "reward": 0.5980728268623352, "reward_std": 0.30698272585868835, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10841288417577744, "sampling/sampling_logp_difference/max": 1.840155839920044, "sampling/importance_sampling_ratio/min": 0.1587926745414734, "sampling/importance_sampling_ratio/mean": 1.000960350036621, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5811756551265717, "clip_ratio/low_mean": 0.03184491116553545, "clip_ratio/low_min": 0.03184491116553545, "clip_ratio/high_mean": 0.05904944706708193, "clip_ratio/high_max": 0.05904944706708193, "clip_ratio/region_mean": 0.09089435823261738, "reward_total_mean": 0.5980728268623352, "reward_meter_mean": 0.6968671679496765, "reward_meter_std": 0.3204905390739441, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9399411678314209, "reward_repeat_soft_std": 0.05086035653948784, "reward_judge_quality_mean": 0.6687500476837158, "reward_judge_quality_std": 0.29844537377357483, "reward_total_composite_mean": 0.5980728268623352, "reward_total_composite_std": 0.30698275566101074} {"timestamp_utc": "2026-04-13T11:07:24Z", "mode": "train", "global_step": 1542, "epoch": 0.15489703666499247, "loss": 0.0187, "grad_norm": 9.514369010925293, "learning_rate": 5.330303030303031e-06, "num_tokens": 2729353.0, "completions/mean_length": 48.75, "completions/min_length": 42.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.75, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.5638735890388489, "rewards/meter/std": 0.3950325548648834, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9393631219863892, "rewards/repeat_soft/std": 0.046571388840675354, "rewards/judge_quality/mean": 0.7325000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.6202842593193054, "rewards/total_composite/std": 0.23236724734306335, "reward": 0.6202842593193054, "reward_std": 0.23236723244190216, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11959965527057648, "sampling/sampling_logp_difference/max": 1.501955509185791, "sampling/importance_sampling_ratio/min": 0.2226942628622055, "sampling/importance_sampling_ratio/mean": 1.0103263854980469, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7916103079915047, "clip_ratio/low_mean": 0.06776149990037084, "clip_ratio/low_min": 0.06776149990037084, "clip_ratio/high_mean": 0.047608437947928905, "clip_ratio/high_max": 0.047608437947928905, "clip_ratio/region_mean": 0.11536993784829974, "reward_total_mean": 0.6202842593193054, "reward_meter_mean": 0.5638735890388489, "reward_meter_std": 0.3950325548648834, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9393631219863892, "reward_repeat_soft_std": 0.046571388840675354, "reward_judge_quality_mean": 0.7325000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.6202842593193054, "reward_total_composite_std": 0.23236724734306335} {"timestamp_utc": "2026-04-13T11:07:32Z", "mode": "train", "global_step": 1543, "epoch": 0.15499748869914615, "loss": -0.0023, "grad_norm": 7.811085224151611, "learning_rate": 5.327272727272727e-06, "num_tokens": 2731860.0, "completions/mean_length": 127.375, "completions/min_length": 100.0, "completions/max_length": 163.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 127.375, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 163.0, "rewards/meter/mean": 0.937529444694519, "rewards/meter/std": 0.07273654639720917, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8321306705474854, "rewards/repeat_soft/std": 0.10501664131879807, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.5340712666511536, "rewards/total_composite/std": 0.03933795914053917, "reward": 0.5340712666511536, "reward_std": 0.03933796286582947, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11424104124307632, "sampling/sampling_logp_difference/max": 1.647552490234375, "sampling/importance_sampling_ratio/min": 0.19252052903175354, "sampling/importance_sampling_ratio/mean": 1.0067178010940552, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.728199802339077, "clip_ratio/low_mean": 0.030051613226532936, "clip_ratio/low_min": 0.030051613226532936, "clip_ratio/high_mean": 0.06726025650277734, "clip_ratio/high_max": 0.06726025650277734, "clip_ratio/region_mean": 0.09731186972931027, "reward_total_mean": 0.5340712666511536, "reward_meter_mean": 0.937529444694519, "reward_meter_std": 0.07273654639720917, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8321306705474854, "reward_repeat_soft_std": 0.10501664131879807, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.5340712666511536, "reward_total_composite_std": 0.03933795914053917} {"timestamp_utc": "2026-04-13T11:07:39Z", "mode": "train", "global_step": 1544, "epoch": 0.15509794073329985, "loss": -0.0121, "grad_norm": 10.400113105773926, "learning_rate": 5.324242424242425e-06, "num_tokens": 2733700.0, "completions/mean_length": 55.0, "completions/min_length": 45.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.0, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.979071855545044, "rewards/meter/std": 0.01928120292723179, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9689692854881287, "rewards/repeat_soft/std": 0.03453250229358673, "rewards/judge_quality/mean": 0.48625001311302185, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.6551904082298279, "rewards/total_composite/std": 0.11491691321134567, "reward": 0.6551904082298279, "reward_std": 0.11491690576076508, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1051703691482544, "sampling/sampling_logp_difference/max": 0.999671459197998, "sampling/importance_sampling_ratio/min": 0.43396344780921936, "sampling/importance_sampling_ratio/mean": 1.0276440382003784, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8735250607132912, "clip_ratio/low_mean": 0.07392397476360202, "clip_ratio/low_min": 0.07392397476360202, "clip_ratio/high_mean": 0.008196720853447914, "clip_ratio/high_max": 0.008196720853447914, "clip_ratio/region_mean": 0.08212069561704993, "reward_total_mean": 0.6551904082298279, "reward_meter_mean": 0.979071855545044, "reward_meter_std": 0.01928120292723179, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9689692854881287, "reward_repeat_soft_std": 0.03453250229358673, "reward_judge_quality_mean": 0.48625001311302185, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.6551904082298279, "reward_total_composite_std": 0.11491691321134567} {"timestamp_utc": "2026-04-13T11:07:46Z", "mode": "train", "global_step": 1545, "epoch": 0.15519839276745354, "loss": 0.008, "grad_norm": 14.160773277282715, "learning_rate": 5.321212121212122e-06, "num_tokens": 2735461.0, "completions/mean_length": 51.125, "completions/min_length": 44.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.125, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.8905959129333496, "rewards/meter/std": 0.2683301270008087, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9754196405410767, "rewards/repeat_soft/std": 0.03474640101194382, "rewards/judge_quality/mean": 0.48625001311302185, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.601094126701355, "rewards/total_composite/std": 0.04728719964623451, "reward": 0.601094126701355, "reward_std": 0.04728719964623451, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1266704797744751, "sampling/sampling_logp_difference/max": 1.1628007888793945, "sampling/importance_sampling_ratio/min": 0.31260940432548523, "sampling/importance_sampling_ratio/mean": 1.0115855932235718, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.953790083527565, "clip_ratio/low_mean": 0.017500000074505806, "clip_ratio/low_min": 0.017500000074505806, "clip_ratio/high_mean": 0.11213839892297983, "clip_ratio/high_max": 0.11213839892297983, "clip_ratio/region_mean": 0.12963839899748564, "reward_total_mean": 0.601094126701355, "reward_meter_mean": 0.8905959129333496, "reward_meter_std": 0.2683301270008087, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9754196405410767, "reward_repeat_soft_std": 0.03474640101194382, "reward_judge_quality_mean": 0.48625001311302185, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.601094126701355, "reward_total_composite_std": 0.04728719964623451} {"timestamp_utc": "2026-04-13T11:07:57Z", "mode": "train", "global_step": 1546, "epoch": 0.15529884480160724, "loss": -0.1924, "grad_norm": 1.910956859588623, "learning_rate": 5.318181818181819e-06, "num_tokens": 2737383.0, "completions/mean_length": 209.25, "completions/min_length": 94.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 108.33333587646484, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.9042237997055054, "rewards/meter/std": 0.13206376135349274, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9281355142593384, "rewards/repeat_soft/std": 0.059728119522333145, "rewards/judge_quality/mean": 0.45500001311302185, "rewards/judge_quality/std": 0.2919393479824066, "rewards/total_composite/mean": 0.5200342535972595, "rewards/total_composite/std": 0.3319006562232971, "reward": 0.5200342535972595, "reward_std": 0.33190062642097473, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11505308002233505, "sampling/sampling_logp_difference/max": 1.2959656715393066, "sampling/importance_sampling_ratio/min": 0.27363350987434387, "sampling/importance_sampling_ratio/mean": 1.0099626779556274, "sampling/importance_sampling_ratio/max": 1.8460681438446045, "entropy": 0.6150470823049545, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.10359776020050049, "clip_ratio/high_max": 0.10359776020050049, "clip_ratio/region_mean": 0.10359776020050049, "reward_total_mean": 0.5200342535972595, "reward_meter_mean": 0.9042237997055054, "reward_meter_std": 0.13206376135349274, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9281355142593384, "reward_repeat_soft_std": 0.059728119522333145, "reward_judge_quality_mean": 0.45500001311302185, "reward_judge_quality_std": 0.2919393479824066, "reward_total_composite_mean": 0.5200342535972595, "reward_total_composite_std": 0.3319006562232971} {"timestamp_utc": "2026-04-13T11:08:05Z", "mode": "train", "global_step": 1547, "epoch": 0.15539929683576093, "loss": 0.0779, "grad_norm": 9.501943588256836, "learning_rate": 5.3151515151515155e-06, "num_tokens": 2739383.0, "completions/mean_length": 75.0, "completions/min_length": 64.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.0, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.9446517825126648, "rewards/meter/std": 0.02080017700791359, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8228054046630859, "rewards/repeat_soft/std": 0.11904467642307281, "rewards/judge_quality/mean": 0.47749999165534973, "rewards/judge_quality/std": 0.2303258776664734, "rewards/total_composite/mean": 0.6034237146377563, "rewards/total_composite/std": 0.15840618312358856, "reward": 0.6034237146377563, "reward_std": 0.15840616822242737, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09276086091995239, "sampling/sampling_logp_difference/max": 2.794069528579712, "sampling/importance_sampling_ratio/min": 0.061171770095825195, "sampling/importance_sampling_ratio/mean": 0.99891197681427, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.36848473735153675, "clip_ratio/low_mean": 0.06293998192995787, "clip_ratio/low_min": 0.06293998192995787, "clip_ratio/high_mean": 0.01844532322138548, "clip_ratio/high_max": 0.01844532322138548, "clip_ratio/region_mean": 0.08138530515134335, "reward_total_mean": 0.6034237146377563, "reward_meter_mean": 0.9446517825126648, "reward_meter_std": 0.02080017700791359, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8228054046630859, "reward_repeat_soft_std": 0.11904467642307281, "reward_judge_quality_mean": 0.47749999165534973, "reward_judge_quality_std": 0.2303258776664734, "reward_total_composite_mean": 0.6034237146377563, "reward_total_composite_std": 0.15840618312358856} {"timestamp_utc": "2026-04-13T11:08:13Z", "mode": "train", "global_step": 1548, "epoch": 0.1554997488699146, "loss": -0.1042, "grad_norm": 7.731827735900879, "learning_rate": 5.312121212121213e-06, "num_tokens": 2741814.0, "completions/mean_length": 112.875, "completions/min_length": 81.0, "completions/max_length": 134.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.875, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 134.0, "rewards/meter/mean": 0.9604154229164124, "rewards/meter/std": 0.015933865681290627, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9142602682113647, "rewards/repeat_soft/std": 0.03433630242943764, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5805824398994446, "rewards/total_composite/std": 0.02216915413737297, "reward": 0.5805824398994446, "reward_std": 0.02216913178563118, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09675724804401398, "sampling/sampling_logp_difference/max": 1.280354380607605, "sampling/importance_sampling_ratio/min": 0.2779387831687927, "sampling/importance_sampling_ratio/mean": 1.0128365755081177, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7276768982410431, "clip_ratio/low_mean": 0.03882122738286853, "clip_ratio/low_min": 0.03882122738286853, "clip_ratio/high_mean": 0.06372740771621466, "clip_ratio/high_max": 0.06372740771621466, "clip_ratio/region_mean": 0.10254863509908319, "reward_total_mean": 0.5805824398994446, "reward_meter_mean": 0.9604154229164124, "reward_meter_std": 0.015933865681290627, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9142602682113647, "reward_repeat_soft_std": 0.03433630242943764, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5805824398994446, "reward_total_composite_std": 0.02216915413737297} {"timestamp_utc": "2026-04-13T11:08:19Z", "mode": "train", "global_step": 1549, "epoch": 0.15560020090406831, "loss": -0.0294, "grad_norm": 15.26572322845459, "learning_rate": 5.309090909090909e-06, "num_tokens": 2743498.0, "completions/mean_length": 39.5, "completions/min_length": 35.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.5, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.8108031749725342, "rewards/meter/std": 0.30466896295547485, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9867522716522217, "rewards/repeat_soft/std": 0.010853597894310951, "rewards/judge_quality/mean": 0.7325000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.7405623197555542, "rewards/total_composite/std": 0.2211495190858841, "reward": 0.7405623197555542, "reward_std": 0.2211495041847229, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1325124204158783, "sampling/sampling_logp_difference/max": 1.735328197479248, "sampling/importance_sampling_ratio/min": 0.17634230852127075, "sampling/importance_sampling_ratio/mean": 0.9997811913490295, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.604987595230341, "clip_ratio/low_mean": 0.053365386091172695, "clip_ratio/low_min": 0.053365386091172695, "clip_ratio/high_mean": 0.05798851721920073, "clip_ratio/high_max": 0.05798851721920073, "clip_ratio/region_mean": 0.11135390331037343, "reward_total_mean": 0.7405623197555542, "reward_meter_mean": 0.8108031749725342, "reward_meter_std": 0.30466896295547485, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9867522716522217, "reward_repeat_soft_std": 0.010853597894310951, "reward_judge_quality_mean": 0.7325000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.7405623197555542, "reward_total_composite_std": 0.2211495190858841} {"timestamp_utc": "2026-04-13T11:08:25Z", "mode": "train", "global_step": 1550, "epoch": 0.155700652938222, "loss": 0.0702, "grad_norm": 17.046966552734375, "learning_rate": 5.306060606060606e-06, "num_tokens": 2744953.0, "completions/mean_length": 18.875, "completions/min_length": 16.0, "completions/max_length": 24.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 18.875, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 24.0, "rewards/meter/mean": 0.9664780497550964, "rewards/meter/std": 0.018068116158246994, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9465569853782654, "rewards/repeat_soft/std": 0.026359153911471367, "rewards/judge_quality/mean": 0.6112500429153442, "rewards/judge_quality/std": 0.25587037205696106, "rewards/total_composite/mean": 0.72638338804245, "rewards/total_composite/std": 0.16048485040664673, "reward": 0.72638338804245, "reward_std": 0.16048482060432434, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11172085255384445, "sampling/sampling_logp_difference/max": 1.3375508785247803, "sampling/importance_sampling_ratio/min": 0.4185125529766083, "sampling/importance_sampling_ratio/mean": 1.0193119049072266, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6745321527123451, "clip_ratio/low_mean": 0.07676024967804551, "clip_ratio/low_min": 0.07676024967804551, "clip_ratio/high_mean": 0.026610644534230232, "clip_ratio/high_max": 0.026610644534230232, "clip_ratio/region_mean": 0.10337089421227574, "reward_total_mean": 0.72638338804245, "reward_meter_mean": 0.9664780497550964, "reward_meter_std": 0.018068116158246994, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9465569853782654, "reward_repeat_soft_std": 0.026359153911471367, "reward_judge_quality_mean": 0.6112500429153442, "reward_judge_quality_std": 0.25587037205696106, "reward_total_composite_mean": 0.72638338804245, "reward_total_composite_std": 0.16048485040664673} {"timestamp_utc": "2026-04-13T11:09:20Z", "mode": "eval", "global_step": 1550, "epoch": 0.155700652938222, "eval_loss": NaN, "eval_runtime": 55.3483, "eval_samples_per_second": 1.445, "eval_steps_per_second": 0.181, "eval_num_tokens": 2744953.0, "eval_completions/mean_length": 97.5375, "eval_completions/min_length": 36.7, "eval_completions/max_length": 238.9, "eval_completions/clipped_ratio": 0.05, "eval_completions/mean_terminated_length": 75.92916793823242, "eval_completions/min_terminated_length": 36.7, "eval_completions/max_terminated_length": 123.3, "eval_rewards/meter/mean": 0.7568637251853942, "eval_rewards/meter/std": 0.29639622159302237, "eval_rewards/count_adherence/mean": 0.9468749761581421, "eval_rewards/count_adherence/std": 0.09409432746469974, "eval_rewards/hard_gate/mean": 0.9375, "eval_rewards/hard_gate/std": 0.15235702097415924, "eval_rewards/repeat_soft/mean": 0.923405921459198, "eval_rewards/repeat_soft/std": 0.06663752272725106, "eval_rewards/judge_quality/mean": 0.5013749897480011, "eval_rewards/judge_quality/std": 0.19865378523245453, "eval_rewards/total_composite/mean": 0.5569140017032623, "eval_rewards/total_composite/std": 0.19040723517537117, "eval_reward": 0.5569140017032623, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.06511233486235142, "eval_sampling/sampling_logp_difference/max": 1.0043030738830567, "eval_sampling/importance_sampling_ratio/min": 0.37784353643655777, "eval_sampling/importance_sampling_ratio/mean": 1.0135634899139405, "eval_sampling/importance_sampling_ratio/max": 1.3928049325942993, "eval_entropy": 0.7222438454627991, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5569140017032623, "eval_reward_meter_mean": 0.7568637251853942, "eval_reward_meter_std": 0.29639622159302237, "eval_reward_count_adherence_mean": 0.9468749761581421, "eval_reward_count_adherence_std": 0.09409432746469974, "eval_reward_hard_gate_mean": 0.9375, "eval_reward_hard_gate_std": 0.15235702097415924, "eval_reward_repeat_soft_mean": 0.923405921459198, "eval_reward_repeat_soft_std": 0.06663752272725106, "eval_reward_judge_quality_mean": 0.5013749897480011, "eval_reward_judge_quality_std": 0.19865378523245453, "eval_reward_total_composite_mean": 0.5569140017032623, "eval_reward_total_composite_std": 0.19040723517537117} {"timestamp_utc": "2026-04-13T11:09:31Z", "mode": "train", "global_step": 1551, "epoch": 0.1558011049723757, "loss": -0.066, "grad_norm": 6.368042469024658, "learning_rate": 5.303030303030303e-06, "num_tokens": 2747835.0, "completions/mean_length": 152.25, "completions/min_length": 118.0, "completions/max_length": 183.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 152.25, "completions/min_terminated_length": 118.0, "completions/max_terminated_length": 183.0, "rewards/meter/mean": 0.9868429899215698, "rewards/meter/std": 0.008497558534145355, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9355205297470093, "rewards/repeat_soft/std": 0.07157598435878754, "rewards/judge_quality/mean": 0.30124998092651367, "rewards/judge_quality/std": 0.10398317128419876, "rewards/total_composite/mean": 0.4304884374141693, "rewards/total_composite/std": 0.18392300605773926, "reward": 0.4304884374141693, "reward_std": 0.18392299115657806, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1302221268415451, "sampling/sampling_logp_difference/max": 1.7571523189544678, "sampling/importance_sampling_ratio/min": 0.1725354939699173, "sampling/importance_sampling_ratio/mean": 1.0175302028656006, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1224376186728477, "clip_ratio/low_mean": 0.026544908992946148, "clip_ratio/low_min": 0.026544908992946148, "clip_ratio/high_mean": 0.099517279304564, "clip_ratio/high_max": 0.099517279304564, "clip_ratio/region_mean": 0.12606218829751015, "reward_total_mean": 0.4304884374141693, "reward_meter_mean": 0.9868429899215698, "reward_meter_std": 0.008497558534145355, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9355205297470093, "reward_repeat_soft_std": 0.07157598435878754, "reward_judge_quality_mean": 0.30124998092651367, "reward_judge_quality_std": 0.10398317128419876, "reward_total_composite_mean": 0.4304884374141693, "reward_total_composite_std": 0.18392300605773926} {"timestamp_utc": "2026-04-13T11:09:37Z", "mode": "train", "global_step": 1552, "epoch": 0.15590155700652938, "loss": -0.0647, "grad_norm": 15.391485214233398, "learning_rate": 5.300000000000001e-06, "num_tokens": 2749242.0, "completions/mean_length": 22.875, "completions/min_length": 15.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.875, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.971284031867981, "rewards/meter/std": 0.01080433838069439, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9125404357910156, "rewards/repeat_soft/std": 0.03231460601091385, "rewards/judge_quality/mean": 0.6737500429153442, "rewards/judge_quality/std": 0.263435423374176, "rewards/total_composite/mean": 0.7617834806442261, "rewards/total_composite/std": 0.16494220495224, "reward": 0.7617834806442261, "reward_std": 0.16494223475456238, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1157231256365776, "sampling/sampling_logp_difference/max": 1.5222516059875488, "sampling/importance_sampling_ratio/min": 0.21821999549865723, "sampling/importance_sampling_ratio/mean": 0.9835056662559509, "sampling/importance_sampling_ratio/max": 1.7103513479232788, "entropy": 0.5968308188021183, "clip_ratio/low_mean": 0.06343860179185867, "clip_ratio/low_min": 0.06343860179185867, "clip_ratio/high_mean": 0.049401901196688414, "clip_ratio/high_max": 0.049401901196688414, "clip_ratio/region_mean": 0.11284050298854709, "reward_total_mean": 0.7617834806442261, "reward_meter_mean": 0.971284031867981, "reward_meter_std": 0.01080433838069439, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9125404357910156, "reward_repeat_soft_std": 0.03231460601091385, "reward_judge_quality_mean": 0.6737500429153442, "reward_judge_quality_std": 0.263435423374176, "reward_total_composite_mean": 0.7617834806442261, "reward_total_composite_std": 0.16494220495224} {"timestamp_utc": "2026-04-13T11:09:44Z", "mode": "train", "global_step": 1553, "epoch": 0.15600200904068307, "loss": 0.0853, "grad_norm": 8.848579406738281, "learning_rate": 5.296969696969697e-06, "num_tokens": 2750795.0, "completions/mean_length": 44.125, "completions/min_length": 34.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.125, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.8539493083953857, "rewards/meter/std": 0.15820306539535522, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9636582136154175, "rewards/repeat_soft/std": 0.04516100138425827, "rewards/judge_quality/mean": 0.48625001311302185, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.6176408529281616, "rewards/total_composite/std": 0.12416830658912659, "reward": 0.6176408529281616, "reward_std": 0.12416831403970718, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12705570459365845, "sampling/sampling_logp_difference/max": 1.7956466674804688, "sampling/importance_sampling_ratio/min": 0.3185770809650421, "sampling/importance_sampling_ratio/mean": 1.0181162357330322, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7664505019783974, "clip_ratio/low_mean": 0.08077190211042762, "clip_ratio/low_min": 0.08077190211042762, "clip_ratio/high_mean": 0.04592503793537617, "clip_ratio/high_max": 0.04592503793537617, "clip_ratio/region_mean": 0.12669694004580379, "reward_total_mean": 0.6176408529281616, "reward_meter_mean": 0.8539493083953857, "reward_meter_std": 0.15820306539535522, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9636582136154175, "reward_repeat_soft_std": 0.04516100138425827, "reward_judge_quality_mean": 0.48625001311302185, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.6176408529281616, "reward_total_composite_std": 0.12416830658912659} {"timestamp_utc": "2026-04-13T11:09:50Z", "mode": "train", "global_step": 1554, "epoch": 0.15610246107483677, "loss": 0.0926, "grad_norm": 18.151506423950195, "learning_rate": 5.293939393939395e-06, "num_tokens": 2752138.0, "completions/mean_length": 23.875, "completions/min_length": 18.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.875, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.893831729888916, "rewards/meter/std": 0.23648913204669952, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9596153497695923, "rewards/repeat_soft/std": 0.008158913813531399, "rewards/judge_quality/mean": 0.7950000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.7982645630836487, "rewards/total_composite/std": 0.17805226147174835, "reward": 0.7982645630836487, "reward_std": 0.17805226147174835, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14744354784488678, "sampling/sampling_logp_difference/max": 1.1399688720703125, "sampling/importance_sampling_ratio/min": 0.31982898712158203, "sampling/importance_sampling_ratio/mean": 1.0191422700881958, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8114384114742279, "clip_ratio/low_mean": 0.029230769723653793, "clip_ratio/low_min": 0.029230769723653793, "clip_ratio/high_mean": 0.0935286795720458, "clip_ratio/high_max": 0.0935286795720458, "clip_ratio/region_mean": 0.1227594492956996, "reward_total_mean": 0.7982645630836487, "reward_meter_mean": 0.893831729888916, "reward_meter_std": 0.23648913204669952, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9596153497695923, "reward_repeat_soft_std": 0.008158913813531399, "reward_judge_quality_mean": 0.7950000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.7982645630836487, "reward_total_composite_std": 0.17805226147174835} {"timestamp_utc": "2026-04-13T11:09:56Z", "mode": "train", "global_step": 1555, "epoch": 0.15620291310899045, "loss": 0.0074, "grad_norm": 10.675823211669922, "learning_rate": 5.290909090909091e-06, "num_tokens": 2754057.0, "completions/mean_length": 73.875, "completions/min_length": 66.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.875, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9280035495758057, "rewards/meter/std": 0.035113625228405, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8458590507507324, "rewards/repeat_soft/std": 0.03096371702849865, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.5825026035308838, "rewards/total_composite/std": 0.013317407108843327, "reward": 0.5825026035308838, "reward_std": 0.01331739965826273, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0960083156824112, "sampling/sampling_logp_difference/max": 1.5728020668029785, "sampling/importance_sampling_ratio/min": 0.20746304094791412, "sampling/importance_sampling_ratio/mean": 0.9991890788078308, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4552772305905819, "clip_ratio/low_mean": 0.033438229002058506, "clip_ratio/low_min": 0.033438229002058506, "clip_ratio/high_mean": 0.0584720391780138, "clip_ratio/high_max": 0.0584720391780138, "clip_ratio/region_mean": 0.09191026818007231, "reward_total_mean": 0.5825026035308838, "reward_meter_mean": 0.9280035495758057, "reward_meter_std": 0.035113625228405, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8458590507507324, "reward_repeat_soft_std": 0.03096371702849865, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.5825026035308838, "reward_total_composite_std": 0.013317407108843327} {"timestamp_utc": "2026-04-13T11:10:04Z", "mode": "train", "global_step": 1556, "epoch": 0.15630336514314414, "loss": 0.0166, "grad_norm": 8.030050277709961, "learning_rate": 5.287878787878788e-06, "num_tokens": 2756457.0, "completions/mean_length": 94.0, "completions/min_length": 76.0, "completions/max_length": 110.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.0, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 110.0, "rewards/meter/mean": 0.8477905988693237, "rewards/meter/std": 0.2628907859325409, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9841978549957275, "rewards/repeat_soft/std": 0.015399984084069729, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.5814843773841858, "rewards/total_composite/std": 0.07323091477155685, "reward": 0.5814843773841858, "reward_std": 0.07323091477155685, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.148122638463974, "sampling/sampling_logp_difference/max": 1.7706518173217773, "sampling/importance_sampling_ratio/min": 0.26092320680618286, "sampling/importance_sampling_ratio/mean": 1.0171198844909668, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1991672441363335, "clip_ratio/low_mean": 0.03827519528567791, "clip_ratio/low_min": 0.03827519528567791, "clip_ratio/high_mean": 0.1132746646180749, "clip_ratio/high_max": 0.1132746646180749, "clip_ratio/region_mean": 0.1515498599037528, "reward_total_mean": 0.5814843773841858, "reward_meter_mean": 0.8477905988693237, "reward_meter_std": 0.2628907859325409, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9841978549957275, "reward_repeat_soft_std": 0.015399984084069729, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.5814843773841858, "reward_total_composite_std": 0.07323091477155685} {"timestamp_utc": "2026-04-13T11:10:16Z", "mode": "train", "global_step": 1557, "epoch": 0.15640381717729784, "loss": -0.1633, "grad_norm": 2.751214027404785, "learning_rate": 5.284848484848485e-06, "num_tokens": 2758656.0, "completions/mean_length": 166.875, "completions/min_length": 97.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 117.5714340209961, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.8048547506332397, "rewards/meter/std": 0.33105307817459106, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9741945862770081, "rewards/repeat_soft/std": 0.017214907333254814, "rewards/judge_quality/mean": 0.39375001192092896, "rewards/judge_quality/std": 0.25048166513442993, "rewards/total_composite/mean": 0.5263358354568481, "rewards/total_composite/std": 0.2428174614906311, "reward": 0.5263358354568481, "reward_std": 0.2428174614906311, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13197213411331177, "sampling/sampling_logp_difference/max": 1.4400482177734375, "sampling/importance_sampling_ratio/min": 0.23691634833812714, "sampling/importance_sampling_ratio/mean": 1.0212165117263794, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8606486544013023, "clip_ratio/low_mean": 0.03507152758538723, "clip_ratio/low_min": 0.03507152758538723, "clip_ratio/high_mean": 0.0820219162851572, "clip_ratio/high_max": 0.0820219162851572, "clip_ratio/region_mean": 0.11709344387054443, "reward_total_mean": 0.5263358354568481, "reward_meter_mean": 0.8048547506332397, "reward_meter_std": 0.33105307817459106, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9741945862770081, "reward_repeat_soft_std": 0.017214907333254814, "reward_judge_quality_mean": 0.39375001192092896, "reward_judge_quality_std": 0.25048166513442993, "reward_total_composite_mean": 0.5263358354568481, "reward_total_composite_std": 0.2428174614906311} {"timestamp_utc": "2026-04-13T11:10:27Z", "mode": "train", "global_step": 1558, "epoch": 0.15650426921145152, "loss": -0.1534, "grad_norm": 2.0736265182495117, "learning_rate": 5.281818181818183e-06, "num_tokens": 2760392.0, "completions/mean_length": 116.0, "completions/min_length": 50.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 59.42857360839844, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.758922815322876, "rewards/meter/std": 0.28375113010406494, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9984496831893921, "rewards/repeat_soft/std": 0.001939329202286899, "rewards/judge_quality/mean": 0.38874998688697815, "rewards/judge_quality/std": 0.13767844438552856, "rewards/total_composite/mean": 0.5132649540901184, "rewards/total_composite/std": 0.213321253657341, "reward": 0.5132649540901184, "reward_std": 0.213321253657341, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15871377289295197, "sampling/sampling_logp_difference/max": 1.6210627555847168, "sampling/importance_sampling_ratio/min": 0.197688490152359, "sampling/importance_sampling_ratio/mean": 1.0191385746002197, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.068819299340248, "clip_ratio/low_mean": 0.01515151560306549, "clip_ratio/low_min": 0.01515151560306549, "clip_ratio/high_mean": 0.09964854456484318, "clip_ratio/high_max": 0.09964854456484318, "clip_ratio/region_mean": 0.11480006016790867, "reward_total_mean": 0.5132649540901184, "reward_meter_mean": 0.758922815322876, "reward_meter_std": 0.28375113010406494, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9984496831893921, "reward_repeat_soft_std": 0.001939329202286899, "reward_judge_quality_mean": 0.38874998688697815, "reward_judge_quality_std": 0.13767844438552856, "reward_total_composite_mean": 0.5132649540901184, "reward_total_composite_std": 0.213321253657341} {"timestamp_utc": "2026-04-13T11:10:33Z", "mode": "train", "global_step": 1559, "epoch": 0.15660472124560523, "loss": -0.0118, "grad_norm": 14.156771659851074, "learning_rate": 5.278787878787879e-06, "num_tokens": 2761903.0, "completions/mean_length": 29.875, "completions/min_length": 26.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.875, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9599217176437378, "rewards/meter/std": 0.06778952479362488, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.958297610282898, "rewards/repeat_soft/std": 0.011176233179867268, "rewards/judge_quality/mean": 0.5637500286102295, "rewards/judge_quality/std": 0.22012579441070557, "rewards/total_composite/mean": 0.6975878477096558, "rewards/total_composite/std": 0.14779505133628845, "reward": 0.6975878477096558, "reward_std": 0.14779506623744965, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09784701466560364, "sampling/sampling_logp_difference/max": 0.8822650909423828, "sampling/importance_sampling_ratio/min": 0.413844496011734, "sampling/importance_sampling_ratio/mean": 1.0062772035598755, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7347541488707066, "clip_ratio/low_mean": 0.059528898447752, "clip_ratio/low_min": 0.059528898447752, "clip_ratio/high_mean": 0.011742424685508013, "clip_ratio/high_max": 0.011742424685508013, "clip_ratio/region_mean": 0.07127132313326001, "reward_total_mean": 0.6975878477096558, "reward_meter_mean": 0.9599217176437378, "reward_meter_std": 0.06778952479362488, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.958297610282898, "reward_repeat_soft_std": 0.011176233179867268, "reward_judge_quality_mean": 0.5637500286102295, "reward_judge_quality_std": 0.22012579441070557, "reward_total_composite_mean": 0.6975878477096558, "reward_total_composite_std": 0.14779505133628845} {"timestamp_utc": "2026-04-13T11:10:45Z", "mode": "train", "global_step": 1560, "epoch": 0.1567051732797589, "loss": -0.1116, "grad_norm": 1.5735626220703125, "learning_rate": 5.2757575757575764e-06, "num_tokens": 2763398.0, "completions/mean_length": 163.875, "completions/min_length": 42.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 47.833335876464844, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.812476396560669, "rewards/meter/std": 0.35773009061813354, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9805666208267212, "rewards/repeat_soft/std": 0.021628832444548607, "rewards/judge_quality/mean": 0.33125001192092896, "rewards/judge_quality/std": 0.1738995909690857, "rewards/total_composite/mean": 0.4497215151786804, "rewards/total_composite/std": 0.28081437945365906, "reward": 0.4497215151786804, "reward_std": 0.28081434965133667, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13447198271751404, "sampling/sampling_logp_difference/max": 2.093869686126709, "sampling/importance_sampling_ratio/min": 0.12320942431688309, "sampling/importance_sampling_ratio/mean": 1.0144320726394653, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6690042838454247, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.079138049390167, "clip_ratio/high_max": 0.079138049390167, "clip_ratio/region_mean": 0.079138049390167, "reward_total_mean": 0.4497215151786804, "reward_meter_mean": 0.812476396560669, "reward_meter_std": 0.35773009061813354, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9805666208267212, "reward_repeat_soft_std": 0.021628832444548607, "reward_judge_quality_mean": 0.33125001192092896, "reward_judge_quality_std": 0.1738995909690857, "reward_total_composite_mean": 0.4497215151786804, "reward_total_composite_std": 0.28081437945365906} {"timestamp_utc": "2026-04-13T11:10:51Z", "mode": "train", "global_step": 1561, "epoch": 0.1568056253139126, "loss": 0.0944, "grad_norm": 12.714040756225586, "learning_rate": 5.272727272727273e-06, "num_tokens": 2765074.0, "completions/mean_length": 49.5, "completions/min_length": 41.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.5, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.6791974306106567, "rewards/meter/std": 0.40748533606529236, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9711135625839233, "rewards/repeat_soft/std": 0.032192979007959366, "rewards/judge_quality/mean": 0.6150000095367432, "rewards/judge_quality/std": 0.25286927819252014, "rewards/total_composite/mean": 0.5897715091705322, "rewards/total_composite/std": 0.15912891924381256, "reward": 0.5897715091705322, "reward_std": 0.15912891924381256, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1287555992603302, "sampling/sampling_logp_difference/max": 1.287729263305664, "sampling/importance_sampling_ratio/min": 0.2758965492248535, "sampling/importance_sampling_ratio/mean": 1.0143460035324097, "sampling/importance_sampling_ratio/max": 1.9852994680404663, "entropy": 1.0453985407948494, "clip_ratio/low_mean": 0.031547619961202145, "clip_ratio/low_min": 0.031547619961202145, "clip_ratio/high_mean": 0.07998319528996944, "clip_ratio/high_max": 0.07998319528996944, "clip_ratio/region_mean": 0.11153081525117159, "reward_total_mean": 0.5897715091705322, "reward_meter_mean": 0.6791974306106567, "reward_meter_std": 0.40748533606529236, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9711135625839233, "reward_repeat_soft_std": 0.032192979007959366, "reward_judge_quality_mean": 0.6150000095367432, "reward_judge_quality_std": 0.25286927819252014, "reward_total_composite_mean": 0.5897715091705322, "reward_total_composite_std": 0.15912891924381256} {"timestamp_utc": "2026-04-13T11:10:59Z", "mode": "train", "global_step": 1562, "epoch": 0.1569060773480663, "loss": 0.0353, "grad_norm": 9.678753852844238, "learning_rate": 5.26969696969697e-06, "num_tokens": 2767338.0, "completions/mean_length": 92.0, "completions/min_length": 82.0, "completions/max_length": 104.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.0, "completions/min_terminated_length": 82.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.845755934715271, "rewards/meter/std": 0.2165696918964386, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8743439316749573, "rewards/repeat_soft/std": 0.13308818638324738, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.5560852289199829, "rewards/total_composite/std": 0.06631311029195786, "reward": 0.5560852289199829, "reward_std": 0.06631311029195786, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13134834170341492, "sampling/sampling_logp_difference/max": 1.8900089263916016, "sampling/importance_sampling_ratio/min": 0.1510704606771469, "sampling/importance_sampling_ratio/mean": 1.0071284770965576, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.544047586619854, "clip_ratio/low_mean": 0.051638985984027386, "clip_ratio/low_min": 0.051638985984027386, "clip_ratio/high_mean": 0.07158447615802288, "clip_ratio/high_max": 0.07158447615802288, "clip_ratio/region_mean": 0.12322346214205027, "reward_total_mean": 0.5560852289199829, "reward_meter_mean": 0.845755934715271, "reward_meter_std": 0.2165696918964386, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8743439316749573, "reward_repeat_soft_std": 0.13308818638324738, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.5560852289199829, "reward_total_composite_std": 0.06631311029195786} {"timestamp_utc": "2026-04-13T11:11:06Z", "mode": "train", "global_step": 1563, "epoch": 0.15700652938221998, "loss": -0.0562, "grad_norm": 6.637635231018066, "learning_rate": 5.2666666666666665e-06, "num_tokens": 2769672.0, "completions/mean_length": 108.75, "completions/min_length": 73.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.75, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.9818673133850098, "rewards/meter/std": 0.013576094061136246, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8238779306411743, "rewards/repeat_soft/std": 0.09231465309858322, "rewards/judge_quality/mean": 0.5562499761581421, "rewards/judge_quality/std": 0.1866959184408188, "rewards/total_composite/mean": 0.6786043047904968, "rewards/total_composite/std": 0.1297779232263565, "reward": 0.6786043047904968, "reward_std": 0.1297779232263565, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08684387803077698, "sampling/sampling_logp_difference/max": 1.6086792945861816, "sampling/importance_sampling_ratio/min": 0.20015178620815277, "sampling/importance_sampling_ratio/mean": 1.0073845386505127, "sampling/importance_sampling_ratio/max": 1.6928097009658813, "entropy": 0.6217380240559578, "clip_ratio/low_mean": 0.03960631275549531, "clip_ratio/low_min": 0.03960631275549531, "clip_ratio/high_mean": 0.02663089195266366, "clip_ratio/high_max": 0.02663089195266366, "clip_ratio/region_mean": 0.06623720470815897, "reward_total_mean": 0.6786043047904968, "reward_meter_mean": 0.9818673133850098, "reward_meter_std": 0.013576094061136246, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8238779306411743, "reward_repeat_soft_std": 0.09231465309858322, "reward_judge_quality_mean": 0.5562499761581421, "reward_judge_quality_std": 0.1866959184408188, "reward_total_composite_mean": 0.6786043047904968, "reward_total_composite_std": 0.1297779232263565} {"timestamp_utc": "2026-04-13T11:11:12Z", "mode": "train", "global_step": 1564, "epoch": 0.1571069814163737, "loss": 0.0165, "grad_norm": 17.923070907592773, "learning_rate": 5.263636363636364e-06, "num_tokens": 2771042.0, "completions/mean_length": 23.25, "completions/min_length": 20.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.25, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9900604486465454, "rewards/meter/std": 0.005437163170427084, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.6170778274536133, "rewards/total_composite/std": 0.007128218654543161, "reward": 0.6170778274536133, "reward_std": 0.007128225173801184, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1269725263118744, "sampling/sampling_logp_difference/max": 1.31215238571167, "sampling/importance_sampling_ratio/min": 0.2692399322986603, "sampling/importance_sampling_ratio/mean": 1.0162959098815918, "sampling/importance_sampling_ratio/max": 1.8745906352996826, "entropy": 0.8843695893883705, "clip_ratio/low_mean": 0.07846467662602663, "clip_ratio/low_min": 0.07846467662602663, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.07846467662602663, "reward_total_mean": 0.6170778274536133, "reward_meter_mean": 0.9900604486465454, "reward_meter_std": 0.005437163170427084, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.6170778274536133, "reward_total_composite_std": 0.007128218654543161} {"timestamp_utc": "2026-04-13T11:11:19Z", "mode": "train", "global_step": 1565, "epoch": 0.15720743345052737, "loss": 0.0515, "grad_norm": 11.058951377868652, "learning_rate": 5.26060606060606e-06, "num_tokens": 2772582.0, "completions/mean_length": 42.5, "completions/min_length": 38.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.5, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.5421852469444275, "rewards/meter/std": 0.4336164891719818, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9924154281616211, "rewards/repeat_soft/std": 0.012711554765701294, "rewards/judge_quality/mean": 0.5600000023841858, "rewards/judge_quality/std": 0.22258226573467255, "rewards/total_composite/mean": 0.5454724431037903, "rewards/total_composite/std": 0.19200962781906128, "reward": 0.5454724431037903, "reward_std": 0.19200964272022247, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13946400582790375, "sampling/sampling_logp_difference/max": 2.6357040405273438, "sampling/importance_sampling_ratio/min": 0.07166849821805954, "sampling/importance_sampling_ratio/mean": 0.9839051365852356, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7409085147082806, "clip_ratio/low_mean": 0.056764266453683376, "clip_ratio/low_min": 0.056764266453683376, "clip_ratio/high_mean": 0.08312970027327538, "clip_ratio/high_max": 0.08312970027327538, "clip_ratio/region_mean": 0.13989396672695875, "reward_total_mean": 0.5454724431037903, "reward_meter_mean": 0.5421852469444275, "reward_meter_std": 0.4336164891719818, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9924154281616211, "reward_repeat_soft_std": 0.012711554765701294, "reward_judge_quality_mean": 0.5600000023841858, "reward_judge_quality_std": 0.22258226573467255, "reward_total_composite_mean": 0.5454724431037903, "reward_total_composite_std": 0.19200962781906128} {"timestamp_utc": "2026-04-13T11:11:25Z", "mode": "train", "global_step": 1566, "epoch": 0.15730788548468105, "loss": 0.0025, "grad_norm": 13.1821870803833, "learning_rate": 5.257575757575758e-06, "num_tokens": 2774094.0, "completions/mean_length": 36.0, "completions/min_length": 32.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.12457047402858734, "rewards/meter/std": 0.33992356061935425, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8819153308868408, "rewards/repeat_soft/std": 0.07343287765979767, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.3662950396537781, "rewards/total_composite/std": 0.0971856340765953, "reward": 0.3662950396537781, "reward_std": 0.0971856340765953, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11318037658929825, "sampling/sampling_logp_difference/max": 1.9890861511230469, "sampling/importance_sampling_ratio/min": 0.13682040572166443, "sampling/importance_sampling_ratio/mean": 0.9863535761833191, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3793261554092169, "clip_ratio/low_mean": 0.0840162024833262, "clip_ratio/low_min": 0.0840162024833262, "clip_ratio/high_mean": 0.01315789483487606, "clip_ratio/high_max": 0.01315789483487606, "clip_ratio/region_mean": 0.09717409731820226, "reward_total_mean": 0.3662950396537781, "reward_meter_mean": 0.12457047402858734, "reward_meter_std": 0.33992356061935425, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8819153308868408, "reward_repeat_soft_std": 0.07343287765979767, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.3662950396537781, "reward_total_composite_std": 0.0971856340765953} {"timestamp_utc": "2026-04-13T11:11:32Z", "mode": "train", "global_step": 1567, "epoch": 0.15740833751883476, "loss": 0.021, "grad_norm": 8.928327560424805, "learning_rate": 5.2545454545454555e-06, "num_tokens": 2775783.0, "completions/mean_length": 59.125, "completions/min_length": 50.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.125, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.8751605749130249, "rewards/meter/std": 0.2470269799232483, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8825733065605164, "rewards/repeat_soft/std": 0.07122129946947098, "rewards/judge_quality/mean": 0.6100000143051147, "rewards/judge_quality/std": 0.24628673493862152, "rewards/total_composite/mean": 0.666850209236145, "rewards/total_composite/std": 0.16365578770637512, "reward": 0.666850209236145, "reward_std": 0.16365578770637512, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11339724808931351, "sampling/sampling_logp_difference/max": 1.3507094383239746, "sampling/importance_sampling_ratio/min": 0.2590563893318176, "sampling/importance_sampling_ratio/mean": 1.0102709531784058, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7823491543531418, "clip_ratio/low_mean": 0.0842600530013442, "clip_ratio/low_min": 0.0842600530013442, "clip_ratio/high_mean": 0.019696970470249653, "clip_ratio/high_max": 0.019696970470249653, "clip_ratio/region_mean": 0.10395702347159386, "reward_total_mean": 0.666850209236145, "reward_meter_mean": 0.8751605749130249, "reward_meter_std": 0.2470269799232483, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8825733065605164, "reward_repeat_soft_std": 0.07122129946947098, "reward_judge_quality_mean": 0.6100000143051147, "reward_judge_quality_std": 0.24628673493862152, "reward_total_composite_mean": 0.666850209236145, "reward_total_composite_std": 0.16365578770637512} {"timestamp_utc": "2026-04-13T11:11:38Z", "mode": "train", "global_step": 1568, "epoch": 0.15750878955298844, "loss": -0.0343, "grad_norm": 13.269099235534668, "learning_rate": 5.251515151515152e-06, "num_tokens": 2777320.0, "completions/mean_length": 47.125, "completions/min_length": 38.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.125, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.7809305191040039, "rewards/meter/std": 0.3151097893714905, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9559616446495056, "rewards/repeat_soft/std": 0.03860804811120033, "rewards/judge_quality/mean": 0.6449999809265137, "rewards/judge_quality/std": 0.24928471446037292, "rewards/total_composite/mean": 0.6395946741104126, "rewards/total_composite/std": 0.13574348390102386, "reward": 0.6395946741104126, "reward_std": 0.13574348390102386, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11881216615438461, "sampling/sampling_logp_difference/max": 1.609349012374878, "sampling/importance_sampling_ratio/min": 0.2000177800655365, "sampling/importance_sampling_ratio/mean": 1.0256962776184082, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6737695634365082, "clip_ratio/low_mean": 0.08250083262100816, "clip_ratio/low_min": 0.08250083262100816, "clip_ratio/high_mean": 0.014433962292969227, "clip_ratio/high_max": 0.014433962292969227, "clip_ratio/region_mean": 0.09693479491397738, "reward_total_mean": 0.6395946741104126, "reward_meter_mean": 0.7809305191040039, "reward_meter_std": 0.3151097893714905, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9559616446495056, "reward_repeat_soft_std": 0.03860804811120033, "reward_judge_quality_mean": 0.6449999809265137, "reward_judge_quality_std": 0.24928471446037292, "reward_total_composite_mean": 0.6395946741104126, "reward_total_composite_std": 0.13574348390102386} {"timestamp_utc": "2026-04-13T11:11:46Z", "mode": "train", "global_step": 1569, "epoch": 0.15760924158714215, "loss": 0.0921, "grad_norm": 19.56976890563965, "learning_rate": 5.248484848484849e-06, "num_tokens": 2779015.0, "completions/mean_length": 38.875, "completions/min_length": 33.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.875, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.8211154341697693, "rewards/meter/std": 0.22690525650978088, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9778926372528076, "rewards/repeat_soft/std": 0.03570332005620003, "rewards/judge_quality/mean": 0.7699999809265137, "rewards/judge_quality/std": 0.22677870094776154, "rewards/total_composite/mean": 0.7675187587738037, "rewards/total_composite/std": 0.17436037957668304, "reward": 0.7675187587738037, "reward_std": 0.17436039447784424, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10905809700489044, "sampling/sampling_logp_difference/max": 1.2931843996047974, "sampling/importance_sampling_ratio/min": 0.27439558506011963, "sampling/importance_sampling_ratio/mean": 1.0121809244155884, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4464311897754669, "clip_ratio/low_mean": 0.024127907119691372, "clip_ratio/low_min": 0.024127907119691372, "clip_ratio/high_mean": 0.05762811517342925, "clip_ratio/high_max": 0.05762811517342925, "clip_ratio/region_mean": 0.08175602229312062, "reward_total_mean": 0.7675187587738037, "reward_meter_mean": 0.8211154341697693, "reward_meter_std": 0.22690525650978088, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9778926372528076, "reward_repeat_soft_std": 0.03570332005620003, "reward_judge_quality_mean": 0.7699999809265137, "reward_judge_quality_std": 0.22677870094776154, "reward_total_composite_mean": 0.7675187587738037, "reward_total_composite_std": 0.17436037957668304} {"timestamp_utc": "2026-04-13T11:11:53Z", "mode": "train", "global_step": 1570, "epoch": 0.15770969362129583, "loss": 0.0308, "grad_norm": 5.472067832946777, "learning_rate": 5.245454545454546e-06, "num_tokens": 2781743.0, "completions/mean_length": 153.0, "completions/min_length": 146.0, "completions/max_length": 166.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 153.0, "completions/min_terminated_length": 146.0, "completions/max_terminated_length": 166.0, "rewards/meter/mean": 0.9894278645515442, "rewards/meter/std": 0.005916774272918701, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7560806274414062, "rewards/repeat_soft/std": 0.0717843696475029, "rewards/judge_quality/mean": 0.29750001430511475, "rewards/judge_quality/std": 0.1349867582321167, "rewards/total_composite/mean": 0.4711803197860718, "rewards/total_composite/std": 0.08733160048723221, "reward": 0.4711803197860718, "reward_std": 0.08733160048723221, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08227641135454178, "sampling/sampling_logp_difference/max": 1.9980177879333496, "sampling/importance_sampling_ratio/min": 0.13560381531715393, "sampling/importance_sampling_ratio/mean": 1.00179123878479, "sampling/importance_sampling_ratio/max": 1.942197322845459, "entropy": 0.4608866088092327, "clip_ratio/low_mean": 0.028170245699584484, "clip_ratio/low_min": 0.028170245699584484, "clip_ratio/high_mean": 0.05055607762187719, "clip_ratio/high_max": 0.05055607762187719, "clip_ratio/region_mean": 0.07872632332146168, "reward_total_mean": 0.4711803197860718, "reward_meter_mean": 0.9894278645515442, "reward_meter_std": 0.005916774272918701, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7560806274414062, "reward_repeat_soft_std": 0.0717843696475029, "reward_judge_quality_mean": 0.29750001430511475, "reward_judge_quality_std": 0.1349867582321167, "reward_total_composite_mean": 0.4711803197860718, "reward_total_composite_std": 0.08733160048723221} {"timestamp_utc": "2026-04-13T11:12:00Z", "mode": "train", "global_step": 1571, "epoch": 0.1578101456554495, "loss": 0.0842, "grad_norm": 9.9489164352417, "learning_rate": 5.242424242424244e-06, "num_tokens": 2783489.0, "completions/mean_length": 56.25, "completions/min_length": 49.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.25, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9855920672416687, "rewards/meter/std": 0.015676230192184448, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9529132843017578, "rewards/repeat_soft/std": 0.04825044795870781, "rewards/judge_quality/mean": 0.5487499833106995, "rewards/judge_quality/std": 0.22937417030334473, "rewards/total_composite/mean": 0.6934705376625061, "rewards/total_composite/std": 0.14567831158638, "reward": 0.6934705376625061, "reward_std": 0.14567831158638, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1211368516087532, "sampling/sampling_logp_difference/max": 1.8575502634048462, "sampling/importance_sampling_ratio/min": 0.15605445206165314, "sampling/importance_sampling_ratio/mean": 1.0127365589141846, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6561549454927444, "clip_ratio/low_mean": 0.08774976572021842, "clip_ratio/low_min": 0.08774976572021842, "clip_ratio/high_mean": 0.028686946723610163, "clip_ratio/high_max": 0.028686946723610163, "clip_ratio/region_mean": 0.11643671244382858, "reward_total_mean": 0.6934705376625061, "reward_meter_mean": 0.9855920672416687, "reward_meter_std": 0.015676230192184448, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9529132843017578, "reward_repeat_soft_std": 0.04825044795870781, "reward_judge_quality_mean": 0.5487499833106995, "reward_judge_quality_std": 0.22937417030334473, "reward_total_composite_mean": 0.6934705376625061, "reward_total_composite_std": 0.14567831158638} {"timestamp_utc": "2026-04-13T11:12:06Z", "mode": "train", "global_step": 1572, "epoch": 0.15791059768960322, "loss": 0.0262, "grad_norm": 10.350753784179688, "learning_rate": 5.23939393939394e-06, "num_tokens": 2785038.0, "completions/mean_length": 35.625, "completions/min_length": 32.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.625, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.9386126399040222, "rewards/meter/std": 0.03209441527724266, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9973484873771667, "rewards/repeat_soft/std": 0.0028345861937850714, "rewards/judge_quality/mean": 0.7987500429153442, "rewards/judge_quality/std": 0.22465451061725616, "rewards/total_composite/mean": 0.8387596011161804, "rewards/total_composite/std": 0.14296871423721313, "reward": 0.8387596011161804, "reward_std": 0.14296871423721313, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0853315219283104, "sampling/sampling_logp_difference/max": 1.6540765762329102, "sampling/importance_sampling_ratio/min": 0.19126859307289124, "sampling/importance_sampling_ratio/mean": 1.0000499486923218, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3170384429395199, "clip_ratio/low_mean": 0.035556891933083534, "clip_ratio/low_min": 0.035556891933083534, "clip_ratio/high_mean": 0.04899440938606858, "clip_ratio/high_max": 0.04899440938606858, "clip_ratio/region_mean": 0.08455130131915212, "reward_total_mean": 0.8387596011161804, "reward_meter_mean": 0.9386126399040222, "reward_meter_std": 0.03209441527724266, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9973484873771667, "reward_repeat_soft_std": 0.0028345861937850714, "reward_judge_quality_mean": 0.7987500429153442, "reward_judge_quality_std": 0.22465451061725616, "reward_total_composite_mean": 0.8387596011161804, "reward_total_composite_std": 0.14296871423721313} {"timestamp_utc": "2026-04-13T11:12:13Z", "mode": "train", "global_step": 1573, "epoch": 0.1580110497237569, "loss": 0.0591, "grad_norm": 11.370959281921387, "learning_rate": 5.236363636363637e-06, "num_tokens": 2786685.0, "completions/mean_length": 55.875, "completions/min_length": 41.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.875, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.8713705539703369, "rewards/meter/std": 0.3254017233848572, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9093286991119385, "rewards/repeat_soft/std": 0.055463191121816635, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.5767042636871338, "rewards/total_composite/std": 0.08499462157487869, "reward": 0.5767042636871338, "reward_std": 0.08499462157487869, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1327207386493683, "sampling/sampling_logp_difference/max": 1.3430163860321045, "sampling/importance_sampling_ratio/min": 0.2610570192337036, "sampling/importance_sampling_ratio/mean": 1.010684609413147, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8427535966038704, "clip_ratio/low_mean": 0.016393441706895828, "clip_ratio/low_min": 0.016393441706895828, "clip_ratio/high_mean": 0.11803002236410975, "clip_ratio/high_max": 0.11803002236410975, "clip_ratio/region_mean": 0.13442346407100558, "reward_total_mean": 0.5767042636871338, "reward_meter_mean": 0.8713705539703369, "reward_meter_std": 0.3254017233848572, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9093286991119385, "reward_repeat_soft_std": 0.055463191121816635, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.5767042636871338, "reward_total_composite_std": 0.08499462157487869} {"timestamp_utc": "2026-04-13T11:12:20Z", "mode": "train", "global_step": 1574, "epoch": 0.1581115017579106, "loss": -0.0634, "grad_norm": 7.636651992797852, "learning_rate": 5.233333333333334e-06, "num_tokens": 2788583.0, "completions/mean_length": 76.25, "completions/min_length": 61.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.25, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9867457151412964, "rewards/meter/std": 0.004665898624807596, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8288445472717285, "rewards/repeat_soft/std": 0.08534655719995499, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5937082767486572, "rewards/total_composite/std": 0.011892632581293583, "reward": 0.5937082767486572, "reward_std": 0.011892640963196754, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11635272949934006, "sampling/sampling_logp_difference/max": 1.3306999206542969, "sampling/importance_sampling_ratio/min": 0.2642922103404999, "sampling/importance_sampling_ratio/mean": 1.0000841617584229, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6212470978498459, "clip_ratio/low_mean": 0.019583360757678747, "clip_ratio/low_min": 0.019583360757678747, "clip_ratio/high_mean": 0.07040644437074661, "clip_ratio/high_max": 0.07040644437074661, "clip_ratio/region_mean": 0.08998980512842536, "reward_total_mean": 0.5937082767486572, "reward_meter_mean": 0.9867457151412964, "reward_meter_std": 0.004665898624807596, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8288445472717285, "reward_repeat_soft_std": 0.08534655719995499, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5937082767486572, "reward_total_composite_std": 0.011892632581293583} {"timestamp_utc": "2026-04-13T11:12:31Z", "mode": "train", "global_step": 1575, "epoch": 0.1582119537920643, "loss": -0.1517, "grad_norm": 3.322378635406494, "learning_rate": 5.230303030303031e-06, "num_tokens": 2790237.0, "completions/mean_length": 120.75, "completions/min_length": 44.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 64.85714721679688, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.5820126533508301, "rewards/meter/std": 0.3395563066005707, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9731205701828003, "rewards/repeat_soft/std": 0.0636206567287445, "rewards/judge_quality/mean": 0.4112499952316284, "rewards/judge_quality/std": 0.1797965168952942, "rewards/total_composite/mean": 0.47881633043289185, "rewards/total_composite/std": 0.22543832659721375, "reward": 0.47881633043289185, "reward_std": 0.22543832659721375, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13704617321491241, "sampling/sampling_logp_difference/max": 1.5153605937957764, "sampling/importance_sampling_ratio/min": 0.21972894668579102, "sampling/importance_sampling_ratio/mean": 1.012619137763977, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5075528435409069, "clip_ratio/low_mean": 0.027435066178441048, "clip_ratio/low_min": 0.027435066178441048, "clip_ratio/high_mean": 0.06846528127789497, "clip_ratio/high_max": 0.06846528127789497, "clip_ratio/region_mean": 0.09590034745633602, "reward_total_mean": 0.47881633043289185, "reward_meter_mean": 0.5820126533508301, "reward_meter_std": 0.3395563066005707, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9731205701828003, "reward_repeat_soft_std": 0.0636206567287445, "reward_judge_quality_mean": 0.4112499952316284, "reward_judge_quality_std": 0.1797965168952942, "reward_total_composite_mean": 0.47881633043289185, "reward_total_composite_std": 0.22543832659721375} {"timestamp_utc": "2026-04-13T11:12:38Z", "mode": "train", "global_step": 1576, "epoch": 0.15831240582621797, "loss": 0.0672, "grad_norm": 8.657578468322754, "learning_rate": 5.2272727272727274e-06, "num_tokens": 2792093.0, "completions/mean_length": 71.0, "completions/min_length": 56.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.0, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.8741507530212402, "rewards/meter/std": 0.13478398323059082, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8293139934539795, "rewards/repeat_soft/std": 0.07133729010820389, "rewards/judge_quality/mean": 0.5612499713897705, "rewards/judge_quality/std": 0.25614938139915466, "rewards/total_composite/mean": 0.6353857517242432, "rewards/total_composite/std": 0.13864047825336456, "reward": 0.6353857517242432, "reward_std": 0.13864046335220337, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0998210459947586, "sampling/sampling_logp_difference/max": 1.2027044296264648, "sampling/importance_sampling_ratio/min": 0.30038073658943176, "sampling/importance_sampling_ratio/mean": 1.0102102756500244, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5485082268714905, "clip_ratio/low_mean": 0.05175823159515858, "clip_ratio/low_min": 0.05175823159515858, "clip_ratio/high_mean": 0.04950087238103151, "clip_ratio/high_max": 0.04950087238103151, "clip_ratio/region_mean": 0.10125910397619009, "reward_total_mean": 0.6353857517242432, "reward_meter_mean": 0.8741507530212402, "reward_meter_std": 0.13478398323059082, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8293139934539795, "reward_repeat_soft_std": 0.07133729010820389, "reward_judge_quality_mean": 0.5612499713897705, "reward_judge_quality_std": 0.25614938139915466, "reward_total_composite_mean": 0.6353857517242432, "reward_total_composite_std": 0.13864047825336456} {"timestamp_utc": "2026-04-13T11:12:45Z", "mode": "train", "global_step": 1577, "epoch": 0.15841285786037168, "loss": 0.0788, "grad_norm": 14.405783653259277, "learning_rate": 5.224242424242425e-06, "num_tokens": 2793539.0, "completions/mean_length": 33.75, "completions/min_length": 28.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.75, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.8508832454681396, "rewards/meter/std": 0.2590540051460266, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9896299839019775, "rewards/repeat_soft/std": 0.010065720416605473, "rewards/judge_quality/mean": 0.6487500667572021, "rewards/judge_quality/std": 0.2507951855659485, "rewards/total_composite/mean": 0.6965362429618835, "rewards/total_composite/std": 0.1784001588821411, "reward": 0.6965362429618835, "reward_std": 0.1784001588821411, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11630921065807343, "sampling/sampling_logp_difference/max": 1.6127252578735352, "sampling/importance_sampling_ratio/min": 0.2402961254119873, "sampling/importance_sampling_ratio/mean": 0.9971534609794617, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5220210142433643, "clip_ratio/low_mean": 0.08428848814219236, "clip_ratio/low_min": 0.08428848814219236, "clip_ratio/high_mean": 0.045522186905145645, "clip_ratio/high_max": 0.045522186905145645, "clip_ratio/region_mean": 0.129810675047338, "reward_total_mean": 0.6965362429618835, "reward_meter_mean": 0.8508832454681396, "reward_meter_std": 0.2590540051460266, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9896299839019775, "reward_repeat_soft_std": 0.010065720416605473, "reward_judge_quality_mean": 0.6487500667572021, "reward_judge_quality_std": 0.2507951855659485, "reward_total_composite_mean": 0.6965362429618835, "reward_total_composite_std": 0.1784001588821411} {"timestamp_utc": "2026-04-13T11:12:57Z", "mode": "train", "global_step": 1578, "epoch": 0.15851330989452536, "loss": -0.0652, "grad_norm": 1.7189676761627197, "learning_rate": 5.221212121212121e-06, "num_tokens": 2795070.0, "completions/mean_length": 210.375, "completions/min_length": 21.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 29.399999618530273, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.19838173687458038, "rewards/meter/std": 0.3413299024105072, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.5349999666213989, "rewards/judge_quality/std": 0.4319060742855072, "rewards/total_composite/mean": 0.28746530413627625, "rewards/total_composite/std": 0.2562018632888794, "reward": 0.28746530413627625, "reward_std": 0.2562018632888794, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16784901916980743, "sampling/sampling_logp_difference/max": 1.1486377716064453, "sampling/importance_sampling_ratio/min": 0.3170683979988098, "sampling/importance_sampling_ratio/mean": 1.0461905002593994, "sampling/importance_sampling_ratio/max": 1.9723833799362183, "entropy": 1.0352696031332016, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.12192688602954149, "clip_ratio/high_max": 0.12192688602954149, "clip_ratio/region_mean": 0.12192688602954149, "reward_total_mean": 0.28746530413627625, "reward_meter_mean": 0.19838173687458038, "reward_meter_std": 0.3413299024105072, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.5349999666213989, "reward_judge_quality_std": 0.4319060742855072, "reward_total_composite_mean": 0.28746530413627625, "reward_total_composite_std": 0.2562018632888794} {"timestamp_utc": "2026-04-13T11:13:04Z", "mode": "train", "global_step": 1579, "epoch": 0.15861376192867904, "loss": -0.0589, "grad_norm": 11.315073013305664, "learning_rate": 5.218181818181819e-06, "num_tokens": 2796880.0, "completions/mean_length": 51.25, "completions/min_length": 41.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.25, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.8555561304092407, "rewards/meter/std": 0.2454449087381363, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.916056752204895, "rewards/repeat_soft/std": 0.06853128224611282, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.5733703374862671, "rewards/total_composite/std": 0.06686627119779587, "reward": 0.5733703374862671, "reward_std": 0.06686627119779587, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1285472810268402, "sampling/sampling_logp_difference/max": 1.6143684387207031, "sampling/importance_sampling_ratio/min": 0.19901631772518158, "sampling/importance_sampling_ratio/mean": 1.0262603759765625, "sampling/importance_sampling_ratio/max": 1.9754222631454468, "entropy": 0.8895671665668488, "clip_ratio/low_mean": 0.029666979797184467, "clip_ratio/low_min": 0.029666979797184467, "clip_ratio/high_mean": 0.09269695077091455, "clip_ratio/high_max": 0.09269695077091455, "clip_ratio/region_mean": 0.12236393056809902, "reward_total_mean": 0.5733703374862671, "reward_meter_mean": 0.8555561304092407, "reward_meter_std": 0.2454449087381363, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.916056752204895, "reward_repeat_soft_std": 0.06853128224611282, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.5733703374862671, "reward_total_composite_std": 0.06686627119779587} {"timestamp_utc": "2026-04-13T11:13:12Z", "mode": "train", "global_step": 1580, "epoch": 0.15871421396283275, "loss": 0.0251, "grad_norm": 7.267446517944336, "learning_rate": 5.215151515151516e-06, "num_tokens": 2799157.0, "completions/mean_length": 117.625, "completions/min_length": 95.0, "completions/max_length": 142.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 117.625, "completions/min_terminated_length": 95.0, "completions/max_terminated_length": 142.0, "rewards/meter/mean": 0.9608144760131836, "rewards/meter/std": 0.05136120691895485, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8403679132461548, "rewards/repeat_soft/std": 0.13496080040931702, "rewards/judge_quality/mean": 0.5900000333786011, "rewards/judge_quality/std": 0.1891333907842636, "rewards/total_composite/mean": 0.6941373348236084, "rewards/total_composite/std": 0.12480224668979645, "reward": 0.6941373348236084, "reward_std": 0.12480224668979645, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1125214546918869, "sampling/sampling_logp_difference/max": 2.707551956176758, "sampling/importance_sampling_ratio/min": 0.06669989228248596, "sampling/importance_sampling_ratio/mean": 1.0147680044174194, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.761542446911335, "clip_ratio/low_mean": 0.039915972389280796, "clip_ratio/low_min": 0.039915972389280796, "clip_ratio/high_mean": 0.047846470028162, "clip_ratio/high_max": 0.047846470028162, "clip_ratio/region_mean": 0.0877624424174428, "reward_total_mean": 0.6941373348236084, "reward_meter_mean": 0.9608144760131836, "reward_meter_std": 0.05136120691895485, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8403679132461548, "reward_repeat_soft_std": 0.13496080040931702, "reward_judge_quality_mean": 0.5900000333786011, "reward_judge_quality_std": 0.1891333907842636, "reward_total_composite_mean": 0.6941373348236084, "reward_total_composite_std": 0.12480224668979645} {"timestamp_utc": "2026-04-13T11:13:18Z", "mode": "train", "global_step": 1581, "epoch": 0.15881466599698643, "loss": 0.0349, "grad_norm": 12.470683097839355, "learning_rate": 5.212121212121213e-06, "num_tokens": 2800876.0, "completions/mean_length": 51.875, "completions/min_length": 46.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.875, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.8088762760162354, "rewards/meter/std": 0.33293840289115906, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9293208122253418, "rewards/repeat_soft/std": 0.01859375834465027, "rewards/judge_quality/mean": 0.5275000333786011, "rewards/judge_quality/std": 0.18873640894889832, "rewards/total_composite/mean": 0.5894694328308105, "rewards/total_composite/std": 0.12230564653873444, "reward": 0.5894694328308105, "reward_std": 0.12230563163757324, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12058217823505402, "sampling/sampling_logp_difference/max": 1.4676432609558105, "sampling/importance_sampling_ratio/min": 0.2304680049419403, "sampling/importance_sampling_ratio/mean": 0.9969631433486938, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7287614792585373, "clip_ratio/low_mean": 0.0363562093116343, "clip_ratio/low_min": 0.0363562093116343, "clip_ratio/high_mean": 0.05789553700014949, "clip_ratio/high_max": 0.05789553700014949, "clip_ratio/region_mean": 0.09425174631178379, "reward_total_mean": 0.5894694328308105, "reward_meter_mean": 0.8088762760162354, "reward_meter_std": 0.33293840289115906, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9293208122253418, "reward_repeat_soft_std": 0.01859375834465027, "reward_judge_quality_mean": 0.5275000333786011, "reward_judge_quality_std": 0.18873640894889832, "reward_total_composite_mean": 0.5894694328308105, "reward_total_composite_std": 0.12230564653873444} {"timestamp_utc": "2026-04-13T11:13:25Z", "mode": "train", "global_step": 1582, "epoch": 0.15891511803114014, "loss": 0.0159, "grad_norm": 17.356124877929688, "learning_rate": 5.209090909090909e-06, "num_tokens": 2802668.0, "completions/mean_length": 63.0, "completions/min_length": 59.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.0, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.7036300301551819, "rewards/meter/std": 0.23615922033786774, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9461240768432617, "rewards/repeat_soft/std": 0.02267386019229889, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.11667262762784958, "rewards/total_composite/mean": 0.5510309338569641, "rewards/total_composite/std": 0.07714087516069412, "reward": 0.5510309338569641, "reward_std": 0.07714088261127472, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08768946677446365, "sampling/sampling_logp_difference/max": 3.278637647628784, "sampling/importance_sampling_ratio/min": 0.037679556757211685, "sampling/importance_sampling_ratio/mean": 0.9985697269439697, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.42118654400110245, "clip_ratio/low_mean": 0.017747920006513596, "clip_ratio/low_min": 0.017747920006513596, "clip_ratio/high_mean": 0.06158871715888381, "clip_ratio/high_max": 0.06158871715888381, "clip_ratio/region_mean": 0.0793366371653974, "reward_total_mean": 0.5510309338569641, "reward_meter_mean": 0.7036300301551819, "reward_meter_std": 0.23615922033786774, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9461240768432617, "reward_repeat_soft_std": 0.02267386019229889, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.11667262762784958, "reward_total_composite_mean": 0.5510309338569641, "reward_total_composite_std": 0.07714087516069412} {"timestamp_utc": "2026-04-13T11:13:32Z", "mode": "train", "global_step": 1583, "epoch": 0.15901557006529382, "loss": -0.0114, "grad_norm": 14.827067375183105, "learning_rate": 5.2060606060606065e-06, "num_tokens": 2804195.0, "completions/mean_length": 35.875, "completions/min_length": 33.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.875, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.7313417792320251, "rewards/meter/std": 0.16144664585590363, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9974443912506104, "rewards/repeat_soft/std": 0.004507945850491524, "rewards/judge_quality/mean": 0.7150000333786011, "rewards/judge_quality/std": 0.24307554960250854, "rewards/total_composite/mean": 0.6837367415428162, "rewards/total_composite/std": 0.12577074766159058, "reward": 0.6837367415428162, "reward_std": 0.12577074766159058, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.144041508436203, "sampling/sampling_logp_difference/max": 3.5102028846740723, "sampling/importance_sampling_ratio/min": 0.029890848323702812, "sampling/importance_sampling_ratio/mean": 0.9953535795211792, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5109865702688694, "clip_ratio/low_mean": 0.047726212767884135, "clip_ratio/low_min": 0.047726212767884135, "clip_ratio/high_mean": 0.057368865702301264, "clip_ratio/high_max": 0.057368865702301264, "clip_ratio/region_mean": 0.1050950784701854, "reward_total_mean": 0.6837367415428162, "reward_meter_mean": 0.7313417792320251, "reward_meter_std": 0.16144664585590363, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9974443912506104, "reward_repeat_soft_std": 0.004507945850491524, "reward_judge_quality_mean": 0.7150000333786011, "reward_judge_quality_std": 0.24307554960250854, "reward_total_composite_mean": 0.6837367415428162, "reward_total_composite_std": 0.12577074766159058} {"timestamp_utc": "2026-04-13T11:13:44Z", "mode": "train", "global_step": 1584, "epoch": 0.1591160220994475, "loss": -0.109, "grad_norm": 3.57348370552063, "learning_rate": 5.203030303030303e-06, "num_tokens": 2806091.0, "completions/mean_length": 142.0, "completions/min_length": 77.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 89.14286041259766, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 115.0, "rewards/meter/mean": 0.7479578256607056, "rewards/meter/std": 0.26813703775405884, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9292437434196472, "rewards/repeat_soft/std": 0.049904964864254, "rewards/judge_quality/mean": 0.6237499713897705, "rewards/judge_quality/std": 0.33907175064086914, "rewards/total_composite/mean": 0.6143718957901001, "rewards/total_composite/std": 0.3079480528831482, "reward": 0.6143718957901001, "reward_std": 0.3079480528831482, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11741182208061218, "sampling/sampling_logp_difference/max": 1.9511277675628662, "sampling/importance_sampling_ratio/min": 0.14211371541023254, "sampling/importance_sampling_ratio/mean": 0.9921196103096008, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.519912701100111, "clip_ratio/low_mean": 0.02894775429740548, "clip_ratio/low_min": 0.02894775429740548, "clip_ratio/high_mean": 0.051714906468987465, "clip_ratio/high_max": 0.051714906468987465, "clip_ratio/region_mean": 0.08066266076639295, "reward_total_mean": 0.6143718957901001, "reward_meter_mean": 0.7479578256607056, "reward_meter_std": 0.26813703775405884, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9292437434196472, "reward_repeat_soft_std": 0.049904964864254, "reward_judge_quality_mean": 0.6237499713897705, "reward_judge_quality_std": 0.33907175064086914, "reward_total_composite_mean": 0.6143718957901001, "reward_total_composite_std": 0.3079480528831482} {"timestamp_utc": "2026-04-13T11:13:51Z", "mode": "train", "global_step": 1585, "epoch": 0.1592164741336012, "loss": 0.1246, "grad_norm": 10.92084789276123, "learning_rate": 5.2e-06, "num_tokens": 2807954.0, "completions/mean_length": 55.875, "completions/min_length": 43.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.875, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.7135234475135803, "rewards/meter/std": 0.3027102053165436, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.930819034576416, "rewards/repeat_soft/std": 0.031038982793688774, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.5591105222702026, "rewards/total_composite/std": 0.09256841987371445, "reward": 0.5591105222702026, "reward_std": 0.09256841987371445, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10486295819282532, "sampling/sampling_logp_difference/max": 1.120133399963379, "sampling/importance_sampling_ratio/min": 0.3262362778186798, "sampling/importance_sampling_ratio/mean": 1.009124517440796, "sampling/importance_sampling_ratio/max": 1.8682174682617188, "entropy": 0.8714019283652306, "clip_ratio/low_mean": 0.0363570312038064, "clip_ratio/low_min": 0.0363570312038064, "clip_ratio/high_mean": 0.06640502624213696, "clip_ratio/high_max": 0.06640502624213696, "clip_ratio/region_mean": 0.10276205744594336, "reward_total_mean": 0.5591105222702026, "reward_meter_mean": 0.7135234475135803, "reward_meter_std": 0.3027102053165436, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.930819034576416, "reward_repeat_soft_std": 0.031038982793688774, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.5591105222702026, "reward_total_composite_std": 0.09256841987371445} {"timestamp_utc": "2026-04-13T11:13:58Z", "mode": "train", "global_step": 1586, "epoch": 0.1593169261677549, "loss": 0.0281, "grad_norm": 14.093966484069824, "learning_rate": 5.196969696969697e-06, "num_tokens": 2809459.0, "completions/mean_length": 45.125, "completions/min_length": 38.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.125, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.7918084859848022, "rewards/meter/std": 0.1963483691215515, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9725749492645264, "rewards/repeat_soft/std": 0.03801165893673897, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.19108712673187256, "rewards/total_composite/mean": 0.5859659910202026, "rewards/total_composite/std": 0.1032167300581932, "reward": 0.5859659910202026, "reward_std": 0.1032167449593544, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1340153068304062, "sampling/sampling_logp_difference/max": 1.883195400238037, "sampling/importance_sampling_ratio/min": 0.15210328996181488, "sampling/importance_sampling_ratio/mean": 1.0184165239334106, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6959806382656097, "clip_ratio/low_mean": 0.020072502084076405, "clip_ratio/low_min": 0.020072502084076405, "clip_ratio/high_mean": 0.08884621039032936, "clip_ratio/high_max": 0.08884621039032936, "clip_ratio/region_mean": 0.10891871247440577, "reward_total_mean": 0.5859659910202026, "reward_meter_mean": 0.7918084859848022, "reward_meter_std": 0.1963483691215515, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9725749492645264, "reward_repeat_soft_std": 0.03801165893673897, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.19108712673187256, "reward_total_composite_mean": 0.5859659910202026, "reward_total_composite_std": 0.1032167300581932} {"timestamp_utc": "2026-04-13T11:14:04Z", "mode": "train", "global_step": 1587, "epoch": 0.1594173782019086, "loss": 0.0598, "grad_norm": 20.771202087402344, "learning_rate": 5.193939393939395e-06, "num_tokens": 2810919.0, "completions/mean_length": 31.5, "completions/min_length": 27.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.5, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.48673421144485474, "rewards/meter/std": 0.358243852853775, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9822884202003479, "rewards/repeat_soft/std": 0.009964286349713802, "rewards/judge_quality/mean": 0.7312500476837158, "rewards/judge_quality/std": 0.20131267607212067, "rewards/total_composite/mean": 0.5827430486679077, "rewards/total_composite/std": 0.19626817107200623, "reward": 0.5827430486679077, "reward_std": 0.19626817107200623, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12723451852798462, "sampling/sampling_logp_difference/max": 2.145318031311035, "sampling/importance_sampling_ratio/min": 0.1170308068394661, "sampling/importance_sampling_ratio/mean": 1.0058783292770386, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4881987236440182, "clip_ratio/low_mean": 0.05622329097241163, "clip_ratio/low_min": 0.05622329097241163, "clip_ratio/high_mean": 0.049179146997630596, "clip_ratio/high_max": 0.049179146997630596, "clip_ratio/region_mean": 0.10540243797004223, "reward_total_mean": 0.5827430486679077, "reward_meter_mean": 0.48673421144485474, "reward_meter_std": 0.358243852853775, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9822884202003479, "reward_repeat_soft_std": 0.009964286349713802, "reward_judge_quality_mean": 0.7312500476837158, "reward_judge_quality_std": 0.20131267607212067, "reward_total_composite_mean": 0.5827430486679077, "reward_total_composite_std": 0.19626817107200623} {"timestamp_utc": "2026-04-13T11:14:15Z", "mode": "train", "global_step": 1588, "epoch": 0.15951783023606228, "loss": -0.0514, "grad_norm": 3.0134294033050537, "learning_rate": 5.190909090909091e-06, "num_tokens": 2812318.0, "completions/mean_length": 83.875, "completions/min_length": 18.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 22.71428680419922, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.701058030128479, "rewards/meter/std": 0.34651628136634827, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9576764702796936, "rewards/repeat_soft/std": 0.013642913661897182, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.3147788345813751, "rewards/total_composite/mean": 0.5050708055496216, "rewards/total_composite/std": 0.26179876923561096, "reward": 0.5050708055496216, "reward_std": 0.2617987394332886, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1175791397690773, "sampling/sampling_logp_difference/max": 1.7134544849395752, "sampling/importance_sampling_ratio/min": 0.18024207651615143, "sampling/importance_sampling_ratio/mean": 0.9964831471443176, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5598728880286217, "clip_ratio/low_mean": 0.054387480951845646, "clip_ratio/low_min": 0.054387480951845646, "clip_ratio/high_mean": 0.06264881044626236, "clip_ratio/high_max": 0.06264881044626236, "clip_ratio/region_mean": 0.117036291398108, "reward_total_mean": 0.5050708055496216, "reward_meter_mean": 0.701058030128479, "reward_meter_std": 0.34651628136634827, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9576764702796936, "reward_repeat_soft_std": 0.013642913661897182, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.3147788345813751, "reward_total_composite_mean": 0.5050708055496216, "reward_total_composite_std": 0.26179876923561096} {"timestamp_utc": "2026-04-13T11:14:21Z", "mode": "train", "global_step": 1589, "epoch": 0.15961828227021596, "loss": 0.0049, "grad_norm": 10.57528305053711, "learning_rate": 5.187878787878788e-06, "num_tokens": 2813869.0, "completions/mean_length": 46.875, "completions/min_length": 35.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.875, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.92644202709198, "rewards/meter/std": 0.16375716030597687, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.908367395401001, "rewards/repeat_soft/std": 0.08913388103246689, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.5940006971359253, "rewards/total_composite/std": 0.05687683820724487, "reward": 0.5940006971359253, "reward_std": 0.05687684565782547, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10283184051513672, "sampling/sampling_logp_difference/max": 1.772202491760254, "sampling/importance_sampling_ratio/min": 0.16995824873447418, "sampling/importance_sampling_ratio/mean": 1.022234320640564, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.686737447977066, "clip_ratio/low_mean": 0.02113095298409462, "clip_ratio/low_min": 0.02113095298409462, "clip_ratio/high_mean": 0.09037395427003503, "clip_ratio/high_max": 0.09037395427003503, "clip_ratio/region_mean": 0.11150490725412965, "reward_total_mean": 0.5940006971359253, "reward_meter_mean": 0.92644202709198, "reward_meter_std": 0.16375716030597687, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.908367395401001, "reward_repeat_soft_std": 0.08913388103246689, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.5940006971359253, "reward_total_composite_std": 0.05687683820724487} {"timestamp_utc": "2026-04-13T11:14:28Z", "mode": "train", "global_step": 1590, "epoch": 0.15971873430436967, "loss": 0.0177, "grad_norm": 6.9513840675354, "learning_rate": 5.184848484848485e-06, "num_tokens": 2815914.0, "completions/mean_length": 96.625, "completions/min_length": 73.0, "completions/max_length": 110.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.625, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 110.0, "rewards/meter/mean": 0.9399340152740479, "rewards/meter/std": 0.0858582854270935, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9564548134803772, "rewards/repeat_soft/std": 0.034047335386276245, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.6397402286529541, "rewards/total_composite/std": 0.1189008429646492, "reward": 0.6397402286529541, "reward_std": 0.118900828063488, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12704916298389435, "sampling/sampling_logp_difference/max": 3.7976233959198, "sampling/importance_sampling_ratio/min": 0.0224240031093359, "sampling/importance_sampling_ratio/mean": 1.0120757818222046, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8992115184664726, "clip_ratio/low_mean": 0.1037961496040225, "clip_ratio/low_min": 0.1037961496040225, "clip_ratio/high_mean": 0.013020833022892475, "clip_ratio/high_max": 0.013020833022892475, "clip_ratio/region_mean": 0.11681698262691498, "reward_total_mean": 0.6397402286529541, "reward_meter_mean": 0.9399340152740479, "reward_meter_std": 0.0858582854270935, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9564548134803772, "reward_repeat_soft_std": 0.034047335386276245, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.6397402286529541, "reward_total_composite_std": 0.1189008429646492} {"timestamp_utc": "2026-04-13T11:14:34Z", "mode": "train", "global_step": 1591, "epoch": 0.15981918633852335, "loss": 0.1216, "grad_norm": 23.791715621948242, "learning_rate": 5.181818181818182e-06, "num_tokens": 2817297.0, "completions/mean_length": 21.875, "completions/min_length": 18.0, "completions/max_length": 25.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.875, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 25.0, "rewards/meter/mean": 0.8926088809967041, "rewards/meter/std": 0.1404341459274292, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.947945773601532, "rewards/repeat_soft/std": 0.04116548225283623, "rewards/judge_quality/mean": 0.29249998927116394, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.511832058429718, "rewards/total_composite/std": 0.05234541743993759, "reward": 0.511832058429718, "reward_std": 0.05234541371464729, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18677814304828644, "sampling/sampling_logp_difference/max": 1.87799072265625, "sampling/importance_sampling_ratio/min": 0.15289701521396637, "sampling/importance_sampling_ratio/mean": 1.034908413887024, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.168588176369667, "clip_ratio/low_mean": 0.12529785558581352, "clip_ratio/low_min": 0.12529785558581352, "clip_ratio/high_mean": 0.054459065198898315, "clip_ratio/high_max": 0.054459065198898315, "clip_ratio/region_mean": 0.17975692078471184, "reward_total_mean": 0.511832058429718, "reward_meter_mean": 0.8926088809967041, "reward_meter_std": 0.1404341459274292, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.947945773601532, "reward_repeat_soft_std": 0.04116548225283623, "reward_judge_quality_mean": 0.29249998927116394, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.511832058429718, "reward_total_composite_std": 0.05234541743993759} {"timestamp_utc": "2026-04-13T11:14:40Z", "mode": "train", "global_step": 1592, "epoch": 0.15991963837267706, "loss": 0.0895, "grad_norm": 15.207016944885254, "learning_rate": 5.1787878787878784e-06, "num_tokens": 2818966.0, "completions/mean_length": 43.625, "completions/min_length": 34.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.625, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.6591077446937561, "rewards/meter/std": 0.35364237427711487, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9206712245941162, "rewards/repeat_soft/std": 0.0638064593076706, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.5571283102035522, "rewards/total_composite/std": 0.17710694670677185, "reward": 0.5571283102035522, "reward_std": 0.17710696160793304, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14776752889156342, "sampling/sampling_logp_difference/max": 2.2035584449768066, "sampling/importance_sampling_ratio/min": 0.11040957272052765, "sampling/importance_sampling_ratio/mean": 0.9959150552749634, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7373897694051266, "clip_ratio/low_mean": 0.04277499159798026, "clip_ratio/low_min": 0.04277499159798026, "clip_ratio/high_mean": 0.07668900489807129, "clip_ratio/high_max": 0.07668900489807129, "clip_ratio/region_mean": 0.11946399649605155, "reward_total_mean": 0.5571283102035522, "reward_meter_mean": 0.6591077446937561, "reward_meter_std": 0.35364237427711487, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9206712245941162, "reward_repeat_soft_std": 0.0638064593076706, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.5571283102035522, "reward_total_composite_std": 0.17710694670677185} {"timestamp_utc": "2026-04-13T11:14:47Z", "mode": "train", "global_step": 1593, "epoch": 0.16002009040683074, "loss": 0.0399, "grad_norm": 10.931285858154297, "learning_rate": 5.1757575757575765e-06, "num_tokens": 2820738.0, "completions/mean_length": 61.5, "completions/min_length": 55.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.5, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.8601760268211365, "rewards/meter/std": 0.2870243191719055, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9424629211425781, "rewards/repeat_soft/std": 0.04623568803071976, "rewards/judge_quality/mean": 0.367499977350235, "rewards/judge_quality/std": 0.09808888286352158, "rewards/total_composite/mean": 0.554111123085022, "rewards/total_composite/std": 0.08999910205602646, "reward": 0.554111123085022, "reward_std": 0.08999911695718765, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12144916504621506, "sampling/sampling_logp_difference/max": 1.7190871238708496, "sampling/importance_sampling_ratio/min": 0.17922969162464142, "sampling/importance_sampling_ratio/mean": 1.0130892992019653, "sampling/importance_sampling_ratio/max": 1.9167232513427734, "entropy": 1.0045983046293259, "clip_ratio/low_mean": 0.061697943136096, "clip_ratio/low_min": 0.061697943136096, "clip_ratio/high_mean": 0.08305913768708706, "clip_ratio/high_max": 0.08305913768708706, "clip_ratio/region_mean": 0.14475708082318306, "reward_total_mean": 0.554111123085022, "reward_meter_mean": 0.8601760268211365, "reward_meter_std": 0.2870243191719055, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9424629211425781, "reward_repeat_soft_std": 0.04623568803071976, "reward_judge_quality_mean": 0.367499977350235, "reward_judge_quality_std": 0.09808888286352158, "reward_total_composite_mean": 0.554111123085022, "reward_total_composite_std": 0.08999910205602646} {"timestamp_utc": "2026-04-13T11:14:56Z", "mode": "train", "global_step": 1594, "epoch": 0.16012054244098442, "loss": 0.0554, "grad_norm": 7.387360572814941, "learning_rate": 5.172727272727273e-06, "num_tokens": 2823131.0, "completions/mean_length": 87.125, "completions/min_length": 80.0, "completions/max_length": 95.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.125, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.8398451209068298, "rewards/meter/std": 0.19838230311870575, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.809610903263092, "rewards/repeat_soft/std": 0.05944375321269035, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5107194185256958, "rewards/total_composite/std": 0.058343540877103806, "reward": 0.5107194185256958, "reward_std": 0.058343540877103806, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09610215574502945, "sampling/sampling_logp_difference/max": 2.1601510047912598, "sampling/importance_sampling_ratio/min": 0.11530770361423492, "sampling/importance_sampling_ratio/mean": 0.9995241761207581, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5531741492450237, "clip_ratio/low_mean": 0.015402622055262327, "clip_ratio/low_min": 0.015402622055262327, "clip_ratio/high_mean": 0.055950433015823364, "clip_ratio/high_max": 0.055950433015823364, "clip_ratio/region_mean": 0.07135305507108569, "reward_total_mean": 0.5107194185256958, "reward_meter_mean": 0.8398451209068298, "reward_meter_std": 0.19838230311870575, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.809610903263092, "reward_repeat_soft_std": 0.05944375321269035, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5107194185256958, "reward_total_composite_std": 0.058343540877103806} {"timestamp_utc": "2026-04-13T11:15:10Z", "mode": "train", "global_step": 1595, "epoch": 0.16022099447513813, "loss": -0.0829, "grad_norm": 1.6510740518569946, "learning_rate": 5.16969696969697e-06, "num_tokens": 2824713.0, "completions/mean_length": 220.75, "completions/min_length": 41.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 46.0, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.21364955604076385, "rewards/meter/std": 0.2237122654914856, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.9871094822883606, "rewards/repeat_soft/std": 0.013966692611575127, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.29605743288993835, "rewards/total_composite/mean": 0.26387202739715576, "rewards/total_composite/std": 0.2245648354291916, "reward": 0.26387202739715576, "reward_std": 0.2245648354291916, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1586504429578781, "sampling/sampling_logp_difference/max": 2.5145673751831055, "sampling/importance_sampling_ratio/min": 0.08089790493249893, "sampling/importance_sampling_ratio/mean": 1.002648949623108, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6664703115820885, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.08654541755095124, "clip_ratio/high_max": 0.08654541755095124, "clip_ratio/region_mean": 0.08654541755095124, "reward_total_mean": 0.26387202739715576, "reward_meter_mean": 0.21364955604076385, "reward_meter_std": 0.2237122654914856, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.9871094822883606, "reward_repeat_soft_std": 0.013966692611575127, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.29605743288993835, "reward_total_composite_mean": 0.26387202739715576, "reward_total_composite_std": 0.2245648354291916} {"timestamp_utc": "2026-04-13T11:15:23Z", "mode": "train", "global_step": 1596, "epoch": 0.1603214465092918, "loss": -0.2111, "grad_norm": 2.712610960006714, "learning_rate": 5.1666666666666675e-06, "num_tokens": 2827335.0, "completions/mean_length": 186.75, "completions/min_length": 123.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 140.2857208251953, "completions/min_terminated_length": 123.0, "completions/max_terminated_length": 157.0, "rewards/meter/mean": 0.6589198112487793, "rewards/meter/std": 0.3036969006061554, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.2357022613286972, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.883997917175293, "rewards/repeat_soft/std": 0.05677222087979317, "rewards/judge_quality/mean": 0.33124998211860657, "rewards/judge_quality/std": 0.13715866208076477, "rewards/total_composite/mean": 0.41331344842910767, "rewards/total_composite/std": 0.1814599186182022, "reward": 0.41331344842910767, "reward_std": 0.18145990371704102, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10325267165899277, "sampling/sampling_logp_difference/max": 1.4933323860168457, "sampling/importance_sampling_ratio/min": 0.22462287545204163, "sampling/importance_sampling_ratio/mean": 1.0112303495407104, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5912267565727234, "clip_ratio/low_mean": 0.01158059504814446, "clip_ratio/low_min": 0.01158059504814446, "clip_ratio/high_mean": 0.07480679452419281, "clip_ratio/high_max": 0.07480679452419281, "clip_ratio/region_mean": 0.08638738957233727, "reward_total_mean": 0.41331344842910767, "reward_meter_mean": 0.6589198112487793, "reward_meter_std": 0.3036969006061554, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.2357022613286972, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.883997917175293, "reward_repeat_soft_std": 0.05677222087979317, "reward_judge_quality_mean": 0.33124998211860657, "reward_judge_quality_std": 0.13715866208076477, "reward_total_composite_mean": 0.41331344842910767, "reward_total_composite_std": 0.1814599186182022} {"timestamp_utc": "2026-04-13T11:15:31Z", "mode": "train", "global_step": 1597, "epoch": 0.16042189854344552, "loss": 0.0247, "grad_norm": 7.101766586303711, "learning_rate": 5.163636363636364e-06, "num_tokens": 2829633.0, "completions/mean_length": 130.25, "completions/min_length": 116.0, "completions/max_length": 161.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 130.25, "completions/min_terminated_length": 116.0, "completions/max_terminated_length": 161.0, "rewards/meter/mean": 0.8372750282287598, "rewards/meter/std": 0.21894672513008118, "rewards/count_adherence/mean": 0.8250000476837158, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9086341857910156, "rewards/repeat_soft/std": 0.07240957766771317, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.49496954679489136, "rewards/total_composite/std": 0.07301773875951767, "reward": 0.49496954679489136, "reward_std": 0.07301773130893707, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11287858337163925, "sampling/sampling_logp_difference/max": 1.9383482933044434, "sampling/importance_sampling_ratio/min": 0.1439415067434311, "sampling/importance_sampling_ratio/mean": 1.019770622253418, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9102844372391701, "clip_ratio/low_mean": 0.03527601668611169, "clip_ratio/low_min": 0.03527601668611169, "clip_ratio/high_mean": 0.060892132110893726, "clip_ratio/high_max": 0.060892132110893726, "clip_ratio/region_mean": 0.09616814879700541, "reward_total_mean": 0.49496954679489136, "reward_meter_mean": 0.8372750282287598, "reward_meter_std": 0.21894672513008118, "reward_count_adherence_mean": 0.8250000476837158, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9086341857910156, "reward_repeat_soft_std": 0.07240957766771317, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.49496954679489136, "reward_total_composite_std": 0.07301773875951767} {"timestamp_utc": "2026-04-13T11:15:39Z", "mode": "train", "global_step": 1598, "epoch": 0.1605223505775992, "loss": -0.0178, "grad_norm": 10.958575248718262, "learning_rate": 5.160606060606061e-06, "num_tokens": 2831096.0, "completions/mean_length": 41.875, "completions/min_length": 34.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.875, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.5208902359008789, "rewards/meter/std": 0.401324987411499, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9967353343963623, "rewards/repeat_soft/std": 0.00640062615275383, "rewards/judge_quality/mean": 0.6349999904632568, "rewards/judge_quality/std": 0.17880557477474213, "rewards/total_composite/mean": 0.5438495874404907, "rewards/total_composite/std": 0.14783990383148193, "reward": 0.5438495874404907, "reward_std": 0.14783990383148193, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14300960302352905, "sampling/sampling_logp_difference/max": 3.099897861480713, "sampling/importance_sampling_ratio/min": 0.04505380615592003, "sampling/importance_sampling_ratio/mean": 1.0001353025436401, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5781061500310898, "clip_ratio/low_mean": 0.05817630235105753, "clip_ratio/low_min": 0.05817630235105753, "clip_ratio/high_mean": 0.06277035549283028, "clip_ratio/high_max": 0.06277035549283028, "clip_ratio/region_mean": 0.1209466578438878, "reward_total_mean": 0.5438495874404907, "reward_meter_mean": 0.5208902359008789, "reward_meter_std": 0.401324987411499, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9967353343963623, "reward_repeat_soft_std": 0.00640062615275383, "reward_judge_quality_mean": 0.6349999904632568, "reward_judge_quality_std": 0.17880557477474213, "reward_total_composite_mean": 0.5438495874404907, "reward_total_composite_std": 0.14783990383148193} {"timestamp_utc": "2026-04-13T11:15:47Z", "mode": "train", "global_step": 1599, "epoch": 0.16062280261175288, "loss": 0.0411, "grad_norm": 11.115921020507812, "learning_rate": 5.1575757575757575e-06, "num_tokens": 2832890.0, "completions/mean_length": 71.25, "completions/min_length": 51.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.25, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.8588839769363403, "rewards/meter/std": 0.20465929806232452, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8969908952713013, "rewards/repeat_soft/std": 0.07345731556415558, "rewards/judge_quality/mean": 0.6362500190734863, "rewards/judge_quality/std": 0.18392062187194824, "rewards/total_composite/mean": 0.6836203336715698, "rewards/total_composite/std": 0.1292240470647812, "reward": 0.6836203336715698, "reward_std": 0.1292240470647812, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12643003463745117, "sampling/sampling_logp_difference/max": 1.3608005046844482, "sampling/importance_sampling_ratio/min": 0.2564553916454315, "sampling/importance_sampling_ratio/mean": 1.0082217454910278, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7264133766293526, "clip_ratio/low_mean": 0.08147330209612846, "clip_ratio/low_min": 0.08147330209612846, "clip_ratio/high_mean": 0.04042944964021444, "clip_ratio/high_max": 0.04042944964021444, "clip_ratio/region_mean": 0.12190275173634291, "reward_total_mean": 0.6836203336715698, "reward_meter_mean": 0.8588839769363403, "reward_meter_std": 0.20465929806232452, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8969908952713013, "reward_repeat_soft_std": 0.07345731556415558, "reward_judge_quality_mean": 0.6362500190734863, "reward_judge_quality_std": 0.18392062187194824, "reward_total_composite_mean": 0.6836203336715698, "reward_total_composite_std": 0.1292240470647812} {"timestamp_utc": "2026-04-13T11:15:58Z", "mode": "train", "global_step": 1600, "epoch": 0.1607232546459066, "loss": -0.1727, "grad_norm": 2.6750028133392334, "learning_rate": 5.154545454545456e-06, "num_tokens": 2835263.0, "completions/mean_length": 164.625, "completions/min_length": 102.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 115.00000762939453, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 127.0, "rewards/meter/mean": 0.812462329864502, "rewards/meter/std": 0.2528509795665741, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9228575229644775, "rewards/repeat_soft/std": 0.06156541034579277, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.2768702805042267, "rewards/total_composite/mean": 0.5253216028213501, "rewards/total_composite/std": 0.23880189657211304, "reward": 0.5253216028213501, "reward_std": 0.23880188167095184, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1308581382036209, "sampling/sampling_logp_difference/max": 2.108630657196045, "sampling/importance_sampling_ratio/min": 0.12140409648418427, "sampling/importance_sampling_ratio/mean": 1.0056959390640259, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7257308885455132, "clip_ratio/low_mean": 0.027334056794643402, "clip_ratio/low_min": 0.027334056794643402, "clip_ratio/high_mean": 0.0635377811267972, "clip_ratio/high_max": 0.0635377811267972, "clip_ratio/region_mean": 0.0908718379214406, "reward_total_mean": 0.5253216028213501, "reward_meter_mean": 0.812462329864502, "reward_meter_std": 0.2528509795665741, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9228575229644775, "reward_repeat_soft_std": 0.06156541034579277, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.2768702805042267, "reward_total_composite_mean": 0.5253216028213501, "reward_total_composite_std": 0.23880189657211304} {"timestamp_utc": "2026-04-13T11:16:55Z", "mode": "eval", "global_step": 1600, "epoch": 0.1607232546459066, "eval_loss": NaN, "eval_runtime": 56.006, "eval_samples_per_second": 1.428, "eval_steps_per_second": 0.179, "eval_num_tokens": 2835263.0, "eval_completions/mean_length": 96.5125, "eval_completions/min_length": 39.7, "eval_completions/max_length": 245.6, "eval_completions/clipped_ratio": 0.0375, "eval_completions/mean_terminated_length": 80.67321548461913, "eval_completions/min_terminated_length": 39.7, "eval_completions/max_terminated_length": 132.1, "eval_rewards/meter/mean": 0.7673405587673188, "eval_rewards/meter/std": 0.268173411488533, "eval_rewards/count_adherence/mean": 0.9447916507720947, "eval_rewards/count_adherence/std": 0.09196658246219158, "eval_rewards/hard_gate/mean": 0.95, "eval_rewards/hard_gate/std": 0.1414213538169861, "eval_rewards/repeat_soft/mean": 0.8982382655143738, "eval_rewards/repeat_soft/std": 0.08441178202629089, "eval_rewards/judge_quality/mean": 0.46950000524520874, "eval_rewards/judge_quality/std": 0.1970653548836708, "eval_rewards/total_composite/mean": 0.5493904024362564, "eval_rewards/total_composite/std": 0.1925140403211117, "eval_reward": 0.5493904024362564, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.062535485252738, "eval_sampling/sampling_logp_difference/max": 1.031826877593994, "eval_sampling/importance_sampling_ratio/min": 0.36239324510097504, "eval_sampling/importance_sampling_ratio/mean": 1.0141837477684021, "eval_sampling/importance_sampling_ratio/max": 1.4294534564018249, "eval_entropy": 0.6990857660770416, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5493904024362564, "eval_reward_meter_mean": 0.7673405587673188, "eval_reward_meter_std": 0.268173411488533, "eval_reward_count_adherence_mean": 0.9447916507720947, "eval_reward_count_adherence_std": 0.09196658246219158, "eval_reward_hard_gate_mean": 0.95, "eval_reward_hard_gate_std": 0.1414213538169861, "eval_reward_repeat_soft_mean": 0.8982382655143738, "eval_reward_repeat_soft_std": 0.08441178202629089, "eval_reward_judge_quality_mean": 0.46950000524520874, "eval_reward_judge_quality_std": 0.1970653548836708, "eval_reward_total_composite_mean": 0.5493904024362564, "eval_reward_total_composite_std": 0.1925140403211117} {"timestamp_utc": "2026-04-13T11:17:09Z", "mode": "train", "global_step": 1601, "epoch": 0.16082370668006027, "loss": -0.17, "grad_norm": 2.645550012588501, "learning_rate": 5.151515151515152e-06, "num_tokens": 2837297.0, "completions/mean_length": 148.25, "completions/min_length": 80.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 96.28572082519531, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.44858697056770325, "rewards/meter/std": 0.30310821533203125, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8310996294021606, "rewards/repeat_soft/std": 0.048754941672086716, "rewards/judge_quality/mean": 0.4775000214576721, "rewards/judge_quality/std": 0.2539263069629669, "rewards/total_composite/mean": 0.4004678726196289, "rewards/total_composite/std": 0.18112020194530487, "reward": 0.4004678726196289, "reward_std": 0.18112018704414368, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10312364995479584, "sampling/sampling_logp_difference/max": 1.91813325881958, "sampling/importance_sampling_ratio/min": 0.1468808948993683, "sampling/importance_sampling_ratio/mean": 0.9989551305770874, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4168057069182396, "clip_ratio/low_mean": 0.027486841194331646, "clip_ratio/low_min": 0.027486841194331646, "clip_ratio/high_mean": 0.05989459529519081, "clip_ratio/high_max": 0.05989459529519081, "clip_ratio/region_mean": 0.08738143648952246, "reward_total_mean": 0.4004678726196289, "reward_meter_mean": 0.44858697056770325, "reward_meter_std": 0.30310821533203125, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8310996294021606, "reward_repeat_soft_std": 0.048754941672086716, "reward_judge_quality_mean": 0.4775000214576721, "reward_judge_quality_std": 0.2539263069629669, "reward_total_composite_mean": 0.4004678726196289, "reward_total_composite_std": 0.18112020194530487} {"timestamp_utc": "2026-04-13T11:17:21Z", "mode": "train", "global_step": 1602, "epoch": 0.16092415871421395, "loss": -0.1398, "grad_norm": 2.3615663051605225, "learning_rate": 5.148484848484849e-06, "num_tokens": 2839115.0, "completions/mean_length": 126.25, "completions/min_length": 62.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 71.14286041259766, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.8981894254684448, "rewards/meter/std": 0.12483187019824982, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9875437021255493, "rewards/repeat_soft/std": 0.012164885178208351, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.22211645543575287, "rewards/total_composite/mean": 0.5541506409645081, "rewards/total_composite/std": 0.2510529160499573, "reward": 0.5541506409645081, "reward_std": 0.2510529160499573, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13300660252571106, "sampling/sampling_logp_difference/max": 2.366598129272461, "sampling/importance_sampling_ratio/min": 0.09379927068948746, "sampling/importance_sampling_ratio/mean": 0.9997671842575073, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8854727745056152, "clip_ratio/low_mean": 0.01689189113676548, "clip_ratio/low_min": 0.01689189113676548, "clip_ratio/high_mean": 0.11515548266470432, "clip_ratio/high_max": 0.11515548266470432, "clip_ratio/region_mean": 0.1320473738014698, "reward_total_mean": 0.5541506409645081, "reward_meter_mean": 0.8981894254684448, "reward_meter_std": 0.12483187019824982, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9875437021255493, "reward_repeat_soft_std": 0.012164885178208351, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.22211645543575287, "reward_total_composite_mean": 0.5541506409645081, "reward_total_composite_std": 0.2510529160499573} {"timestamp_utc": "2026-04-13T11:17:27Z", "mode": "train", "global_step": 1603, "epoch": 0.16102461074836766, "loss": 0.0289, "grad_norm": 11.913492202758789, "learning_rate": 5.145454545454546e-06, "num_tokens": 2840919.0, "completions/mean_length": 43.5, "completions/min_length": 36.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.5, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.8211958408355713, "rewards/meter/std": 0.2598826587200165, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9115462899208069, "rewards/repeat_soft/std": 0.05229616537690163, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.22403764724731445, "rewards/total_composite/mean": 0.5429308414459229, "rewards/total_composite/std": 0.07537403702735901, "reward": 0.5429308414459229, "reward_std": 0.07537402212619781, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11811105161905289, "sampling/sampling_logp_difference/max": 1.1230762004852295, "sampling/importance_sampling_ratio/min": 0.3252776265144348, "sampling/importance_sampling_ratio/mean": 1.0160083770751953, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8078811019659042, "clip_ratio/low_mean": 0.05214684922248125, "clip_ratio/low_min": 0.05214684922248125, "clip_ratio/high_mean": 0.08267963957041502, "clip_ratio/high_max": 0.08267963957041502, "clip_ratio/region_mean": 0.13482648879289627, "reward_total_mean": 0.5429308414459229, "reward_meter_mean": 0.8211958408355713, "reward_meter_std": 0.2598826587200165, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9115462899208069, "reward_repeat_soft_std": 0.05229616537690163, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.22403764724731445, "reward_total_composite_mean": 0.5429308414459229, "reward_total_composite_std": 0.07537403702735901} {"timestamp_utc": "2026-04-13T11:17:34Z", "mode": "train", "global_step": 1604, "epoch": 0.16112506278252134, "loss": 0.0867, "grad_norm": 18.628082275390625, "learning_rate": 5.142424242424243e-06, "num_tokens": 2842277.0, "completions/mean_length": 24.75, "completions/min_length": 22.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.75, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.8188531398773193, "rewards/meter/std": 0.34643879532814026, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.53125, "rewards/judge_quality/std": 0.24833375215530396, "rewards/total_composite/mean": 0.6387529373168945, "rewards/total_composite/std": 0.20481234788894653, "reward": 0.6387529373168945, "reward_std": 0.20481234788894653, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12999440729618073, "sampling/sampling_logp_difference/max": 1.6062629222869873, "sampling/importance_sampling_ratio/min": 0.20063599944114685, "sampling/importance_sampling_ratio/mean": 1.0326086282730103, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9076575338840485, "clip_ratio/low_mean": 0.08458688016980886, "clip_ratio/low_min": 0.08458688016980886, "clip_ratio/high_mean": 0.03840909153223038, "clip_ratio/high_max": 0.03840909153223038, "clip_ratio/region_mean": 0.12299597170203924, "reward_total_mean": 0.6387529373168945, "reward_meter_mean": 0.8188531398773193, "reward_meter_std": 0.34643879532814026, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.53125, "reward_judge_quality_std": 0.24833375215530396, "reward_total_composite_mean": 0.6387529373168945, "reward_total_composite_std": 0.20481234788894653} {"timestamp_utc": "2026-04-13T11:17:41Z", "mode": "train", "global_step": 1605, "epoch": 0.16122551481667505, "loss": 0.0535, "grad_norm": 16.2254638671875, "learning_rate": 5.139393939393939e-06, "num_tokens": 2843690.0, "completions/mean_length": 22.625, "completions/min_length": 18.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.625, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.9037623405456543, "rewards/meter/std": 0.17803955078125, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.3462499976158142, "rewards/judge_quality/std": 0.10336308926343918, "rewards/total_composite/mean": 0.5511980056762695, "rewards/total_composite/std": 0.07901346683502197, "reward": 0.5511980056762695, "reward_std": 0.07901346683502197, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13475042581558228, "sampling/sampling_logp_difference/max": 1.7557687759399414, "sampling/importance_sampling_ratio/min": 0.17277435958385468, "sampling/importance_sampling_ratio/mean": 1.0087201595306396, "sampling/importance_sampling_ratio/max": 1.7569578886032104, "entropy": 0.8755036517977715, "clip_ratio/low_mean": 0.022361111361533403, "clip_ratio/low_min": 0.022361111361533403, "clip_ratio/high_mean": 0.06772394385188818, "clip_ratio/high_max": 0.06772394385188818, "clip_ratio/region_mean": 0.09008505521342158, "reward_total_mean": 0.5511980056762695, "reward_meter_mean": 0.9037623405456543, "reward_meter_std": 0.17803955078125, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.3462499976158142, "reward_judge_quality_std": 0.10336308926343918, "reward_total_composite_mean": 0.5511980056762695, "reward_total_composite_std": 0.07901346683502197} {"timestamp_utc": "2026-04-13T11:17:47Z", "mode": "train", "global_step": 1606, "epoch": 0.16132596685082873, "loss": -0.0313, "grad_norm": 14.100306510925293, "learning_rate": 5.1363636363636375e-06, "num_tokens": 2845067.0, "completions/mean_length": 22.125, "completions/min_length": 18.0, "completions/max_length": 24.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.125, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 24.0, "rewards/meter/mean": 0.9127197265625, "rewards/meter/std": 0.09622737020254135, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9573202729225159, "rewards/repeat_soft/std": 0.011277626268565655, "rewards/judge_quality/mean": 0.45625001192092896, "rewards/judge_quality/std": 0.21185828745365143, "rewards/total_composite/mean": 0.6148068904876709, "rewards/total_composite/std": 0.12625357508659363, "reward": 0.6148068904876709, "reward_std": 0.12625358998775482, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1196405440568924, "sampling/sampling_logp_difference/max": 1.4739208221435547, "sampling/importance_sampling_ratio/min": 0.22902576625347137, "sampling/importance_sampling_ratio/mean": 1.0227696895599365, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7830843478441238, "clip_ratio/low_mean": 0.08167270477861166, "clip_ratio/low_min": 0.08167270477861166, "clip_ratio/high_mean": 0.03645833395421505, "clip_ratio/high_max": 0.03645833395421505, "clip_ratio/region_mean": 0.11813103873282671, "reward_total_mean": 0.6148068904876709, "reward_meter_mean": 0.9127197265625, "reward_meter_std": 0.09622737020254135, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9573202729225159, "reward_repeat_soft_std": 0.011277626268565655, "reward_judge_quality_mean": 0.45625001192092896, "reward_judge_quality_std": 0.21185828745365143, "reward_total_composite_mean": 0.6148068904876709, "reward_total_composite_std": 0.12625357508659363} {"timestamp_utc": "2026-04-13T11:17:53Z", "mode": "train", "global_step": 1607, "epoch": 0.1614264188849824, "loss": 0.0546, "grad_norm": 13.587140083312988, "learning_rate": 5.133333333333334e-06, "num_tokens": 2846618.0, "completions/mean_length": 40.875, "completions/min_length": 37.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.875, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.6240348815917969, "rewards/meter/std": 0.4144648611545563, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9673728942871094, "rewards/repeat_soft/std": 0.04772808402776718, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.605128288269043, "rewards/total_composite/std": 0.2106853872537613, "reward": 0.605128288269043, "reward_std": 0.2106853872537613, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10679075866937637, "sampling/sampling_logp_difference/max": 3.874640464782715, "sampling/importance_sampling_ratio/min": 0.020761799067258835, "sampling/importance_sampling_ratio/mean": 1.0191521644592285, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5109239295125008, "clip_ratio/low_mean": 0.07677730452269316, "clip_ratio/low_min": 0.07677730452269316, "clip_ratio/high_mean": 0.03319597104564309, "clip_ratio/high_max": 0.03319597104564309, "clip_ratio/region_mean": 0.10997327556833625, "reward_total_mean": 0.605128288269043, "reward_meter_mean": 0.6240348815917969, "reward_meter_std": 0.4144648611545563, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9673728942871094, "reward_repeat_soft_std": 0.04772808402776718, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.605128288269043, "reward_total_composite_std": 0.2106853872537613} {"timestamp_utc": "2026-04-13T11:18:01Z", "mode": "train", "global_step": 1608, "epoch": 0.16152687091913612, "loss": -0.0536, "grad_norm": 7.547384262084961, "learning_rate": 5.130303030303031e-06, "num_tokens": 2849068.0, "completions/mean_length": 109.25, "completions/min_length": 92.0, "completions/max_length": 130.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 109.25, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 130.0, "rewards/meter/mean": 0.9808225631713867, "rewards/meter/std": 0.0065976474434137344, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.841166615486145, "rewards/repeat_soft/std": 0.06930014491081238, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.5404136776924133, "rewards/total_composite/std": 0.0432850681245327, "reward": 0.5404136776924133, "reward_std": 0.043285053223371506, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10502967983484268, "sampling/sampling_logp_difference/max": 1.924008846282959, "sampling/importance_sampling_ratio/min": 0.14602041244506836, "sampling/importance_sampling_ratio/mean": 1.0074819326400757, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7072211876511574, "clip_ratio/low_mean": 0.020380434580147266, "clip_ratio/low_min": 0.020380434580147266, "clip_ratio/high_mean": 0.07942765299230814, "clip_ratio/high_max": 0.07942765299230814, "clip_ratio/region_mean": 0.0998080875724554, "reward_total_mean": 0.5404136776924133, "reward_meter_mean": 0.9808225631713867, "reward_meter_std": 0.0065976474434137344, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.841166615486145, "reward_repeat_soft_std": 0.06930014491081238, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.5404136776924133, "reward_total_composite_std": 0.0432850681245327} {"timestamp_utc": "2026-04-13T11:18:12Z", "mode": "train", "global_step": 1609, "epoch": 0.1616273229532898, "loss": -0.0007, "grad_norm": 16.270160675048828, "learning_rate": 5.1272727272727275e-06, "num_tokens": 2850776.0, "completions/mean_length": 47.5, "completions/min_length": 37.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.5, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.847195029258728, "rewards/meter/std": 0.323220819234848, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9017714262008667, "rewards/repeat_soft/std": 0.030768852680921555, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.5402442216873169, "rewards/total_composite/std": 0.09381219744682312, "reward": 0.5402442216873169, "reward_std": 0.09381220489740372, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10080137848854065, "sampling/sampling_logp_difference/max": 1.389388084411621, "sampling/importance_sampling_ratio/min": 0.24922776222229004, "sampling/importance_sampling_ratio/mean": 0.9930405616760254, "sampling/importance_sampling_ratio/max": 1.710292935371399, "entropy": 0.6024736911058426, "clip_ratio/low_mean": 0.039260704070329666, "clip_ratio/low_min": 0.039260704070329666, "clip_ratio/high_mean": 0.03794161765836179, "clip_ratio/high_max": 0.03794161765836179, "clip_ratio/region_mean": 0.07720232172869146, "reward_total_mean": 0.5402442216873169, "reward_meter_mean": 0.847195029258728, "reward_meter_std": 0.323220819234848, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9017714262008667, "reward_repeat_soft_std": 0.030768852680921555, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.5402442216873169, "reward_total_composite_std": 0.09381219744682312} {"timestamp_utc": "2026-04-13T11:18:23Z", "mode": "train", "global_step": 1610, "epoch": 0.1617277749874435, "loss": -0.1903, "grad_norm": 2.445298433303833, "learning_rate": 5.124242424242425e-06, "num_tokens": 2853113.0, "completions/mean_length": 186.125, "completions/min_length": 127.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 139.57144165039062, "completions/min_terminated_length": 127.0, "completions/max_terminated_length": 148.0, "rewards/meter/mean": 0.8654612302780151, "rewards/meter/std": 0.17720764875411987, "rewards/count_adherence/mean": 0.8541666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8106361031532288, "rewards/repeat_soft/std": 0.029637936502695084, "rewards/judge_quality/mean": 0.41499999165534973, "rewards/judge_quality/std": 0.21633309125900269, "rewards/total_composite/mean": 0.4939092993736267, "rewards/total_composite/std": 0.229624405503273, "reward": 0.4939092993736267, "reward_std": 0.22962439060211182, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09812058508396149, "sampling/sampling_logp_difference/max": 1.952284812927246, "sampling/importance_sampling_ratio/min": 0.1419493705034256, "sampling/importance_sampling_ratio/mean": 1.0083082914352417, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.561351902782917, "clip_ratio/low_mean": 0.02542989421635866, "clip_ratio/low_min": 0.02542989421635866, "clip_ratio/high_mean": 0.05852258764207363, "clip_ratio/high_max": 0.05852258764207363, "clip_ratio/region_mean": 0.08395248185843229, "reward_total_mean": 0.4939092993736267, "reward_meter_mean": 0.8654612302780151, "reward_meter_std": 0.17720764875411987, "reward_count_adherence_mean": 0.8541666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8106361031532288, "reward_repeat_soft_std": 0.029637936502695084, "reward_judge_quality_mean": 0.41499999165534973, "reward_judge_quality_std": 0.21633309125900269, "reward_total_composite_mean": 0.4939092993736267, "reward_total_composite_std": 0.229624405503273} {"timestamp_utc": "2026-04-13T11:18:29Z", "mode": "train", "global_step": 1611, "epoch": 0.1618282270215972, "loss": 0.0264, "grad_norm": 12.777203559875488, "learning_rate": 5.121212121212121e-06, "num_tokens": 2854664.0, "completions/mean_length": 42.875, "completions/min_length": 39.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.875, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.6671066284179688, "rewards/meter/std": 0.336922287940979, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9815101623535156, "rewards/repeat_soft/std": 0.018132537603378296, "rewards/judge_quality/mean": 0.7362500429153442, "rewards/judge_quality/std": 0.25376805663108826, "rewards/total_composite/mean": 0.6651303768157959, "rewards/total_composite/std": 0.18651041388511658, "reward": 0.6651303768157959, "reward_std": 0.18651039898395538, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10066421329975128, "sampling/sampling_logp_difference/max": 1.44620943069458, "sampling/importance_sampling_ratio/min": 0.23546114563941956, "sampling/importance_sampling_ratio/mean": 0.9972047209739685, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.44500717893242836, "clip_ratio/low_mean": 0.03842564998194575, "clip_ratio/low_min": 0.03842564998194575, "clip_ratio/high_mean": 0.04087729286402464, "clip_ratio/high_max": 0.04087729286402464, "clip_ratio/region_mean": 0.07930294284597039, "reward_total_mean": 0.6651303768157959, "reward_meter_mean": 0.6671066284179688, "reward_meter_std": 0.336922287940979, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9815101623535156, "reward_repeat_soft_std": 0.018132537603378296, "reward_judge_quality_mean": 0.7362500429153442, "reward_judge_quality_std": 0.25376805663108826, "reward_total_composite_mean": 0.6651303768157959, "reward_total_composite_std": 0.18651041388511658} {"timestamp_utc": "2026-04-13T11:18:36Z", "mode": "train", "global_step": 1612, "epoch": 0.16192867905575087, "loss": 0.0258, "grad_norm": 7.783813953399658, "learning_rate": 5.1181818181818185e-06, "num_tokens": 2857219.0, "completions/mean_length": 114.375, "completions/min_length": 104.0, "completions/max_length": 140.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 114.375, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 140.0, "rewards/meter/mean": 0.9686040282249451, "rewards/meter/std": 0.022422077134251595, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8075989484786987, "rewards/repeat_soft/std": 0.07894428819417953, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.6254061460494995, "rewards/total_composite/std": 0.11612702906131744, "reward": 0.6254061460494995, "reward_std": 0.11612702161073685, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09219850599765778, "sampling/sampling_logp_difference/max": 1.4588334560394287, "sampling/importance_sampling_ratio/min": 0.2325073480606079, "sampling/importance_sampling_ratio/mean": 1.004338264465332, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6646230891346931, "clip_ratio/low_mean": 0.06814689747989178, "clip_ratio/low_min": 0.06814689747989178, "clip_ratio/high_mean": 0.009259259328246117, "clip_ratio/high_max": 0.009259259328246117, "clip_ratio/region_mean": 0.0774061568081379, "reward_total_mean": 0.6254061460494995, "reward_meter_mean": 0.9686040282249451, "reward_meter_std": 0.022422077134251595, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8075989484786987, "reward_repeat_soft_std": 0.07894428819417953, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.6254061460494995, "reward_total_composite_std": 0.11612702906131744} {"timestamp_utc": "2026-04-13T11:18:49Z", "mode": "train", "global_step": 1613, "epoch": 0.16202913108990458, "loss": -0.2401, "grad_norm": 1.867802381515503, "learning_rate": 5.115151515151515e-06, "num_tokens": 2859937.0, "completions/mean_length": 199.75, "completions/min_length": 127.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 155.1428680419922, "completions/min_terminated_length": 127.0, "completions/max_terminated_length": 192.0, "rewards/meter/mean": 0.9049348831176758, "rewards/meter/std": 0.14540007710456848, "rewards/count_adherence/mean": 0.8500000238418579, "rewards/count_adherence/std": 0.1414213478565216, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8620929718017578, "rewards/repeat_soft/std": 0.07246813923120499, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.48327773809432983, "rewards/total_composite/std": 0.20073343813419342, "reward": 0.48327773809432983, "reward_std": 0.20073342323303223, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10677609592676163, "sampling/sampling_logp_difference/max": 2.683267593383789, "sampling/importance_sampling_ratio/min": 0.06833948194980621, "sampling/importance_sampling_ratio/mean": 1.0043482780456543, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5648591257631779, "clip_ratio/low_mean": 0.012867647223174572, "clip_ratio/low_min": 0.012867647223174572, "clip_ratio/high_mean": 0.06916146818548441, "clip_ratio/high_max": 0.06916146818548441, "clip_ratio/region_mean": 0.08202911540865898, "reward_total_mean": 0.48327773809432983, "reward_meter_mean": 0.9049348831176758, "reward_meter_std": 0.14540007710456848, "reward_count_adherence_mean": 0.8500000238418579, "reward_count_adherence_std": 0.1414213478565216, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8620929718017578, "reward_repeat_soft_std": 0.07246813923120499, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.48327773809432983, "reward_total_composite_std": 0.20073343813419342} {"timestamp_utc": "2026-04-13T11:19:00Z", "mode": "train", "global_step": 1614, "epoch": 0.16212958312405826, "loss": -0.1467, "grad_norm": 2.9070706367492676, "learning_rate": 5.112121212121213e-06, "num_tokens": 2861901.0, "completions/mean_length": 130.5, "completions/min_length": 68.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 76.0, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.8117458820343018, "rewards/meter/std": 0.31275635957717896, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.7944717407226562, "rewards/repeat_soft/std": 0.10366769880056381, "rewards/judge_quality/mean": 0.33125001192092896, "rewards/judge_quality/std": 0.13715866208076477, "rewards/total_composite/mean": 0.46274441480636597, "rewards/total_composite/std": 0.21137595176696777, "reward": 0.46274441480636597, "reward_std": 0.21137595176696777, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08003520220518112, "sampling/sampling_logp_difference/max": 1.8128626346588135, "sampling/importance_sampling_ratio/min": 0.16318632662296295, "sampling/importance_sampling_ratio/mean": 1.011260986328125, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.34130437672138214, "clip_ratio/low_mean": 0.004687500186264515, "clip_ratio/low_min": 0.004687500186264515, "clip_ratio/high_mean": 0.05632254155352712, "clip_ratio/high_max": 0.05632254155352712, "clip_ratio/region_mean": 0.06101004173979163, "reward_total_mean": 0.46274441480636597, "reward_meter_mean": 0.8117458820343018, "reward_meter_std": 0.31275635957717896, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.7944717407226562, "reward_repeat_soft_std": 0.10366769880056381, "reward_judge_quality_mean": 0.33125001192092896, "reward_judge_quality_std": 0.13715866208076477, "reward_total_composite_mean": 0.46274441480636597, "reward_total_composite_std": 0.21137595176696777} {"timestamp_utc": "2026-04-13T11:19:06Z", "mode": "train", "global_step": 1615, "epoch": 0.16223003515821197, "loss": 0.0854, "grad_norm": 10.08071517944336, "learning_rate": 5.109090909090909e-06, "num_tokens": 2863527.0, "completions/mean_length": 50.25, "completions/min_length": 43.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.25, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.8818612694740295, "rewards/meter/std": 0.27754804491996765, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8982758522033691, "rewards/repeat_soft/std": 0.07960784435272217, "rewards/judge_quality/mean": 0.606249988079071, "rewards/judge_quality/std": 0.2492811232805252, "rewards/total_composite/mean": 0.693257212638855, "rewards/total_composite/std": 0.19573436677455902, "reward": 0.693257212638855, "reward_std": 0.19573435187339783, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11168952286243439, "sampling/sampling_logp_difference/max": 1.2561886310577393, "sampling/importance_sampling_ratio/min": 0.2847371995449066, "sampling/importance_sampling_ratio/mean": 1.0139323472976685, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7719178572297096, "clip_ratio/low_mean": 0.053172151325270534, "clip_ratio/low_min": 0.053172151325270534, "clip_ratio/high_mean": 0.045388901606202126, "clip_ratio/high_max": 0.045388901606202126, "clip_ratio/region_mean": 0.09856105293147266, "reward_total_mean": 0.693257212638855, "reward_meter_mean": 0.8818612694740295, "reward_meter_std": 0.27754804491996765, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8982758522033691, "reward_repeat_soft_std": 0.07960784435272217, "reward_judge_quality_mean": 0.606249988079071, "reward_judge_quality_std": 0.2492811232805252, "reward_total_composite_mean": 0.693257212638855, "reward_total_composite_std": 0.19573436677455902} {"timestamp_utc": "2026-04-13T11:19:12Z", "mode": "train", "global_step": 1616, "epoch": 0.16233048719236565, "loss": 0.095, "grad_norm": 19.3624210357666, "learning_rate": 5.106060606060607e-06, "num_tokens": 2864983.0, "completions/mean_length": 25.0, "completions/min_length": 17.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.0, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9213307499885559, "rewards/meter/std": 0.14470230042934418, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9494390487670898, "rewards/repeat_soft/std": 0.03694179654121399, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.5803236961364746, "rewards/total_composite/std": 0.04910682886838913, "reward": 0.5803236961364746, "reward_std": 0.04910682141780853, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14111049473285675, "sampling/sampling_logp_difference/max": 1.9327747821807861, "sampling/importance_sampling_ratio/min": 0.14474600553512573, "sampling/importance_sampling_ratio/mean": 1.0063116550445557, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9546639770269394, "clip_ratio/low_mean": 0.047727273777127266, "clip_ratio/low_min": 0.047727273777127266, "clip_ratio/high_mean": 0.1064573572948575, "clip_ratio/high_max": 0.1064573572948575, "clip_ratio/region_mean": 0.15418463107198477, "reward_total_mean": 0.5803236961364746, "reward_meter_mean": 0.9213307499885559, "reward_meter_std": 0.14470230042934418, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9494390487670898, "reward_repeat_soft_std": 0.03694179654121399, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.5803236961364746, "reward_total_composite_std": 0.04910682886838913} {"timestamp_utc": "2026-04-13T11:19:19Z", "mode": "train", "global_step": 1617, "epoch": 0.16243093922651933, "loss": 0.0395, "grad_norm": 9.595121383666992, "learning_rate": 5.103030303030303e-06, "num_tokens": 2866961.0, "completions/mean_length": 71.25, "completions/min_length": 60.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.25, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.9378806352615356, "rewards/meter/std": 0.08389707654714584, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.956716775894165, "rewards/repeat_soft/std": 0.005239794030785561, "rewards/judge_quality/mean": 0.6775000095367432, "rewards/judge_quality/std": 0.15645630657672882, "rewards/total_composite/mean": 0.7527035474777222, "rewards/total_composite/std": 0.08143587410449982, "reward": 0.7527035474777222, "reward_std": 0.08143586665391922, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10567019134759903, "sampling/sampling_logp_difference/max": 1.5408134460449219, "sampling/importance_sampling_ratio/min": 0.21420679986476898, "sampling/importance_sampling_ratio/mean": 1.0093913078308105, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5745541453361511, "clip_ratio/low_mean": 0.01658957451581955, "clip_ratio/low_min": 0.01658957451581955, "clip_ratio/high_mean": 0.07150601642206311, "clip_ratio/high_max": 0.07150601642206311, "clip_ratio/region_mean": 0.08809559093788266, "reward_total_mean": 0.7527035474777222, "reward_meter_mean": 0.9378806352615356, "reward_meter_std": 0.08389707654714584, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.956716775894165, "reward_repeat_soft_std": 0.005239794030785561, "reward_judge_quality_mean": 0.6775000095367432, "reward_judge_quality_std": 0.15645630657672882, "reward_total_composite_mean": 0.7527035474777222, "reward_total_composite_std": 0.08143587410449982} {"timestamp_utc": "2026-04-13T11:19:25Z", "mode": "train", "global_step": 1618, "epoch": 0.16253139126067304, "loss": 0.0556, "grad_norm": 13.132983207702637, "learning_rate": 5.1e-06, "num_tokens": 2868520.0, "completions/mean_length": 35.875, "completions/min_length": 33.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.875, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.7930529713630676, "rewards/meter/std": 0.27568158507347107, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9989027976989746, "rewards/repeat_soft/std": 0.002089538611471653, "rewards/judge_quality/mean": 0.6112500429153442, "rewards/judge_quality/std": 0.25587037205696106, "rewards/total_composite/mean": 0.6624631285667419, "rewards/total_composite/std": 0.17252354323863983, "reward": 0.6624631285667419, "reward_std": 0.17252352833747864, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08907654136419296, "sampling/sampling_logp_difference/max": 1.5596189498901367, "sampling/importance_sampling_ratio/min": 0.4622865915298462, "sampling/importance_sampling_ratio/mean": 1.020411491394043, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3689921572804451, "clip_ratio/low_mean": 0.06342792743816972, "clip_ratio/low_min": 0.06342792743816972, "clip_ratio/high_mean": 0.029411765281111002, "clip_ratio/high_max": 0.029411765281111002, "clip_ratio/region_mean": 0.09283969271928072, "reward_total_mean": 0.6624631285667419, "reward_meter_mean": 0.7930529713630676, "reward_meter_std": 0.27568158507347107, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9989027976989746, "reward_repeat_soft_std": 0.002089538611471653, "reward_judge_quality_mean": 0.6112500429153442, "reward_judge_quality_std": 0.25587037205696106, "reward_total_composite_mean": 0.6624631285667419, "reward_total_composite_std": 0.17252354323863983} {"timestamp_utc": "2026-04-13T11:19:36Z", "mode": "train", "global_step": 1619, "epoch": 0.16263184329482672, "loss": -0.1018, "grad_norm": 3.055025815963745, "learning_rate": 5.096969696969697e-06, "num_tokens": 2870943.0, "completions/mean_length": 188.875, "completions/min_length": 102.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 142.71429443359375, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 188.0, "rewards/meter/mean": 0.3641549348831177, "rewards/meter/std": 0.39495429396629333, "rewards/count_adherence/mean": 0.8541666269302368, "rewards/count_adherence/std": 0.2878147065639496, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8099063038825989, "rewards/repeat_soft/std": 0.1071389839053154, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.34355688095092773, "rewards/total_composite/std": 0.17706337571144104, "reward": 0.34355688095092773, "reward_std": 0.17706336081027985, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1074066087603569, "sampling/sampling_logp_difference/max": 1.994025707244873, "sampling/importance_sampling_ratio/min": 0.13614623248577118, "sampling/importance_sampling_ratio/mean": 1.004380464553833, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4460967816412449, "clip_ratio/low_mean": 0.046347555704414845, "clip_ratio/low_min": 0.046347555704414845, "clip_ratio/high_mean": 0.029849925078451633, "clip_ratio/high_max": 0.029849925078451633, "clip_ratio/region_mean": 0.07619748078286648, "reward_total_mean": 0.34355688095092773, "reward_meter_mean": 0.3641549348831177, "reward_meter_std": 0.39495429396629333, "reward_count_adherence_mean": 0.8541666269302368, "reward_count_adherence_std": 0.2878147065639496, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8099063038825989, "reward_repeat_soft_std": 0.1071389839053154, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.34355688095092773, "reward_total_composite_std": 0.17706337571144104} {"timestamp_utc": "2026-04-13T11:19:44Z", "mode": "train", "global_step": 1620, "epoch": 0.16273229532898043, "loss": 0.0153, "grad_norm": 7.13424825668335, "learning_rate": 5.093939393939395e-06, "num_tokens": 2873349.0, "completions/mean_length": 121.75, "completions/min_length": 105.0, "completions/max_length": 135.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 121.75, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 135.0, "rewards/meter/mean": 0.7869986295700073, "rewards/meter/std": 0.22102504968643188, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8619891405105591, "rewards/repeat_soft/std": 0.059215281158685684, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.5751908421516418, "rewards/total_composite/std": 0.13902568817138672, "reward": 0.5751908421516418, "reward_std": 0.13902568817138672, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12312452495098114, "sampling/sampling_logp_difference/max": 2.1598167419433594, "sampling/importance_sampling_ratio/min": 0.115346260368824, "sampling/importance_sampling_ratio/mean": 1.0205926895141602, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0005771592259407, "clip_ratio/low_mean": 0.04885080549865961, "clip_ratio/low_min": 0.04885080549865961, "clip_ratio/high_mean": 0.06665956694632769, "clip_ratio/high_max": 0.06665956694632769, "clip_ratio/region_mean": 0.1155103724449873, "reward_total_mean": 0.5751908421516418, "reward_meter_mean": 0.7869986295700073, "reward_meter_std": 0.22102504968643188, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8619891405105591, "reward_repeat_soft_std": 0.059215281158685684, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.5751908421516418, "reward_total_composite_std": 0.13902568817138672} {"timestamp_utc": "2026-04-13T11:19:57Z", "mode": "train", "global_step": 1621, "epoch": 0.1628327473631341, "loss": -0.0467, "grad_norm": 4.196609973907471, "learning_rate": 5.090909090909091e-06, "num_tokens": 2874984.0, "completions/mean_length": 104.375, "completions/min_length": 32.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 46.142860412597656, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.4812016189098358, "rewards/meter/std": 0.37489452958106995, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9685986042022705, "rewards/repeat_soft/std": 0.06977908313274384, "rewards/judge_quality/mean": 0.5612500309944153, "rewards/judge_quality/std": 0.3223324716091156, "rewards/total_composite/mean": 0.43485894799232483, "rewards/total_composite/std": 0.23783420026302338, "reward": 0.43485894799232483, "reward_std": 0.23783420026302338, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12444452941417694, "sampling/sampling_logp_difference/max": 2.1760754585266113, "sampling/importance_sampling_ratio/min": 0.11348603665828705, "sampling/importance_sampling_ratio/mean": 1.0035669803619385, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4616983085870743, "clip_ratio/low_mean": 0.037504821084439754, "clip_ratio/low_min": 0.037504821084439754, "clip_ratio/high_mean": 0.06014752900227904, "clip_ratio/high_max": 0.06014752900227904, "clip_ratio/region_mean": 0.0976523500867188, "reward_total_mean": 0.43485894799232483, "reward_meter_mean": 0.4812016189098358, "reward_meter_std": 0.37489452958106995, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9685986042022705, "reward_repeat_soft_std": 0.06977908313274384, "reward_judge_quality_mean": 0.5612500309944153, "reward_judge_quality_std": 0.3223324716091156, "reward_total_composite_mean": 0.43485894799232483, "reward_total_composite_std": 0.23783420026302338} {"timestamp_utc": "2026-04-13T11:20:03Z", "mode": "train", "global_step": 1622, "epoch": 0.1629331993972878, "loss": -0.0033, "grad_norm": 12.808685302734375, "learning_rate": 5.0878787878787885e-06, "num_tokens": 2876610.0, "completions/mean_length": 42.25, "completions/min_length": 37.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.25, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.761579155921936, "rewards/meter/std": 0.2876737117767334, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9501906633377075, "rewards/repeat_soft/std": 0.05366018787026405, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.5522565841674805, "rewards/total_composite/std": 0.07785069197416306, "reward": 0.5522565841674805, "reward_std": 0.07785066962242126, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11096014082431793, "sampling/sampling_logp_difference/max": 1.0162148475646973, "sampling/importance_sampling_ratio/min": 0.3619624376296997, "sampling/importance_sampling_ratio/mean": 0.996402382850647, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5768076628446579, "clip_ratio/low_mean": 0.0435085401404649, "clip_ratio/low_min": 0.0435085401404649, "clip_ratio/high_mean": 0.0699709472246468, "clip_ratio/high_max": 0.0699709472246468, "clip_ratio/region_mean": 0.11347948736511171, "reward_total_mean": 0.5522565841674805, "reward_meter_mean": 0.761579155921936, "reward_meter_std": 0.2876737117767334, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9501906633377075, "reward_repeat_soft_std": 0.05366018787026405, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.5522565841674805, "reward_total_composite_std": 0.07785069197416306} {"timestamp_utc": "2026-04-13T11:20:09Z", "mode": "train", "global_step": 1623, "epoch": 0.1630336514314415, "loss": 0.0678, "grad_norm": 17.057676315307617, "learning_rate": 5.084848484848486e-06, "num_tokens": 2878072.0, "completions/mean_length": 35.75, "completions/min_length": 31.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.75, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.5909867882728577, "rewards/meter/std": 0.37435105443000793, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9783687591552734, "rewards/repeat_soft/std": 0.03977498412132263, "rewards/judge_quality/mean": 0.5487500429153442, "rewards/judge_quality/std": 0.22937415540218353, "rewards/total_composite/mean": 0.542860746383667, "rewards/total_composite/std": 0.14172787964344025, "reward": 0.542860746383667, "reward_std": 0.14172787964344025, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1391218900680542, "sampling/sampling_logp_difference/max": 2.116826057434082, "sampling/importance_sampling_ratio/min": 0.12041321396827698, "sampling/importance_sampling_ratio/mean": 0.9980522990226746, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47563890740275383, "clip_ratio/low_mean": 0.04050394566729665, "clip_ratio/low_min": 0.04050394566729665, "clip_ratio/high_mean": 0.09049043618142605, "clip_ratio/high_max": 0.09049043618142605, "clip_ratio/region_mean": 0.1309943818487227, "reward_total_mean": 0.542860746383667, "reward_meter_mean": 0.5909867882728577, "reward_meter_std": 0.37435105443000793, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9783687591552734, "reward_repeat_soft_std": 0.03977498412132263, "reward_judge_quality_mean": 0.5487500429153442, "reward_judge_quality_std": 0.22937415540218353, "reward_total_composite_mean": 0.542860746383667, "reward_total_composite_std": 0.14172787964344025} {"timestamp_utc": "2026-04-13T11:20:16Z", "mode": "train", "global_step": 1624, "epoch": 0.16313410346559518, "loss": -0.0129, "grad_norm": 8.933954238891602, "learning_rate": 5.081818181818182e-06, "num_tokens": 2880331.0, "completions/mean_length": 87.375, "completions/min_length": 69.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.375, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.8871557712554932, "rewards/meter/std": 0.16150468587875366, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.828206479549408, "rewards/repeat_soft/std": 0.04936762526631355, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.10507651418447495, "rewards/total_composite/mean": 0.5923798084259033, "rewards/total_composite/std": 0.08415987342596054, "reward": 0.5923798084259033, "reward_std": 0.08415987342596054, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1125749796628952, "sampling/sampling_logp_difference/max": 1.4348459243774414, "sampling/importance_sampling_ratio/min": 0.23815205693244934, "sampling/importance_sampling_ratio/mean": 1.0075618028640747, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7525682523846626, "clip_ratio/low_mean": 0.03880079975351691, "clip_ratio/low_min": 0.03880079975351691, "clip_ratio/high_mean": 0.05439908429980278, "clip_ratio/high_max": 0.05439908429980278, "clip_ratio/region_mean": 0.09319988405331969, "reward_total_mean": 0.5923798084259033, "reward_meter_mean": 0.8871557712554932, "reward_meter_std": 0.16150468587875366, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.828206479549408, "reward_repeat_soft_std": 0.04936762526631355, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.10507651418447495, "reward_total_composite_mean": 0.5923798084259033, "reward_total_composite_std": 0.08415987342596054} {"timestamp_utc": "2026-04-13T11:20:22Z", "mode": "train", "global_step": 1625, "epoch": 0.16323455549974886, "loss": 0.0353, "grad_norm": 16.366573333740234, "learning_rate": 5.078787878787879e-06, "num_tokens": 2881690.0, "completions/mean_length": 21.875, "completions/min_length": 19.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.875, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.7903445959091187, "rewards/meter/std": 0.32402628660202026, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.937935471534729, "rewards/repeat_soft/std": 0.040868453681468964, "rewards/judge_quality/mean": 0.42124998569488525, "rewards/judge_quality/std": 0.06998724490404129, "rewards/total_composite/mean": 0.5541468262672424, "rewards/total_composite/std": 0.09281747043132782, "reward": 0.5541468262672424, "reward_std": 0.09281745553016663, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.144802987575531, "sampling/sampling_logp_difference/max": 1.0754953622817993, "sampling/importance_sampling_ratio/min": 0.3917113244533539, "sampling/importance_sampling_ratio/mean": 1.0188369750976562, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7806518524885178, "clip_ratio/low_mean": 0.05853242985904217, "clip_ratio/low_min": 0.05853242985904217, "clip_ratio/high_mean": 0.08434872049838305, "clip_ratio/high_max": 0.08434872049838305, "clip_ratio/region_mean": 0.1428811503574252, "reward_total_mean": 0.5541468262672424, "reward_meter_mean": 0.7903445959091187, "reward_meter_std": 0.32402628660202026, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.937935471534729, "reward_repeat_soft_std": 0.040868453681468964, "reward_judge_quality_mean": 0.42124998569488525, "reward_judge_quality_std": 0.06998724490404129, "reward_total_composite_mean": 0.5541468262672424, "reward_total_composite_std": 0.09281747043132782} {"timestamp_utc": "2026-04-13T11:20:29Z", "mode": "train", "global_step": 1626, "epoch": 0.16333500753390257, "loss": -0.014, "grad_norm": 9.989459991455078, "learning_rate": 5.075757575757576e-06, "num_tokens": 2883911.0, "completions/mean_length": 103.625, "completions/min_length": 84.0, "completions/max_length": 130.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 103.625, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 130.0, "rewards/meter/mean": 0.778885543346405, "rewards/meter/std": 0.38644787669181824, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6627370119094849, "rewards/repeat_soft/std": 0.16214919090270996, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.49848833680152893, "rewards/total_composite/std": 0.09143819659948349, "reward": 0.49848833680152893, "reward_std": 0.0914381891489029, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08699291944503784, "sampling/sampling_logp_difference/max": 1.5512089729309082, "sampling/importance_sampling_ratio/min": 0.21199151873588562, "sampling/importance_sampling_ratio/mean": 0.9883591532707214, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.42546793818473816, "clip_ratio/low_mean": 0.03153377631679177, "clip_ratio/low_min": 0.03153377631679177, "clip_ratio/high_mean": 0.05089415982365608, "clip_ratio/high_max": 0.05089415982365608, "clip_ratio/region_mean": 0.08242793614044785, "reward_total_mean": 0.49848833680152893, "reward_meter_mean": 0.778885543346405, "reward_meter_std": 0.38644787669181824, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6627370119094849, "reward_repeat_soft_std": 0.16214919090270996, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.49848833680152893, "reward_total_composite_std": 0.09143819659948349} {"timestamp_utc": "2026-04-13T11:20:36Z", "mode": "train", "global_step": 1627, "epoch": 0.16343545956805625, "loss": 0.0291, "grad_norm": 11.057270050048828, "learning_rate": 5.072727272727274e-06, "num_tokens": 2885676.0, "completions/mean_length": 50.625, "completions/min_length": 40.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.625, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9232890605926514, "rewards/meter/std": 0.0828256607055664, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8278992176055908, "rewards/repeat_soft/std": 0.11741218715906143, "rewards/judge_quality/mean": 0.5349999666213989, "rewards/judge_quality/std": 0.24663449823856354, "rewards/total_composite/mean": 0.6471283435821533, "rewards/total_composite/std": 0.16611239314079285, "reward": 0.6471283435821533, "reward_std": 0.16611239314079285, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10562446713447571, "sampling/sampling_logp_difference/max": 1.053572654724121, "sampling/importance_sampling_ratio/min": 0.3486897647380829, "sampling/importance_sampling_ratio/mean": 1.019388198852539, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7362676560878754, "clip_ratio/low_mean": 0.07240128144621849, "clip_ratio/low_min": 0.07240128144621849, "clip_ratio/high_mean": 0.02590811997652054, "clip_ratio/high_max": 0.02590811997652054, "clip_ratio/region_mean": 0.09830940142273903, "reward_total_mean": 0.6471283435821533, "reward_meter_mean": 0.9232890605926514, "reward_meter_std": 0.0828256607055664, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8278992176055908, "reward_repeat_soft_std": 0.11741218715906143, "reward_judge_quality_mean": 0.5349999666213989, "reward_judge_quality_std": 0.24663449823856354, "reward_total_composite_mean": 0.6471283435821533, "reward_total_composite_std": 0.16611239314079285} {"timestamp_utc": "2026-04-13T11:20:42Z", "mode": "train", "global_step": 1628, "epoch": 0.16353591160220995, "loss": -0.0076, "grad_norm": 11.232089042663574, "learning_rate": 5.06969696969697e-06, "num_tokens": 2887311.0, "completions/mean_length": 58.375, "completions/min_length": 47.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.375, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9384300708770752, "rewards/meter/std": 0.12097222357988358, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9826440811157227, "rewards/repeat_soft/std": 0.014948301948606968, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.13695022463798523, "rewards/total_composite/mean": 0.6538193225860596, "rewards/total_composite/std": 0.101405069231987, "reward": 0.6538193225860596, "reward_std": 0.1014050617814064, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12784811854362488, "sampling/sampling_logp_difference/max": 1.5097036361694336, "sampling/importance_sampling_ratio/min": 0.22097545862197876, "sampling/importance_sampling_ratio/mean": 1.0171589851379395, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8741703927516937, "clip_ratio/low_mean": 0.09202880971133709, "clip_ratio/low_min": 0.09202880971133709, "clip_ratio/high_mean": 0.02007575798779726, "clip_ratio/high_max": 0.02007575798779726, "clip_ratio/region_mean": 0.11210456769913435, "reward_total_mean": 0.6538193225860596, "reward_meter_mean": 0.9384300708770752, "reward_meter_std": 0.12097222357988358, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9826440811157227, "reward_repeat_soft_std": 0.014948301948606968, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.13695022463798523, "reward_total_composite_mean": 0.6538193225860596, "reward_total_composite_std": 0.101405069231987} {"timestamp_utc": "2026-04-13T11:20:49Z", "mode": "train", "global_step": 1629, "epoch": 0.16363636363636364, "loss": 0.0336, "grad_norm": 12.065169334411621, "learning_rate": 5.0666666666666676e-06, "num_tokens": 2888937.0, "completions/mean_length": 52.25, "completions/min_length": 48.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.25, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9526774883270264, "rewards/meter/std": 0.07780593633651733, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9728760123252869, "rewards/repeat_soft/std": 0.0253890473395586, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.6060123443603516, "rewards/total_composite/std": 0.022973021492362022, "reward": 0.6060123443603516, "reward_std": 0.022973012179136276, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13202910125255585, "sampling/sampling_logp_difference/max": 1.6118357181549072, "sampling/importance_sampling_ratio/min": 0.199521005153656, "sampling/importance_sampling_ratio/mean": 1.0097827911376953, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.658332109451294, "clip_ratio/low_mean": 0.028846154920756817, "clip_ratio/low_min": 0.028846154920756817, "clip_ratio/high_mean": 0.09457175945863128, "clip_ratio/high_max": 0.09457175945863128, "clip_ratio/region_mean": 0.1234179143793881, "reward_total_mean": 0.6060123443603516, "reward_meter_mean": 0.9526774883270264, "reward_meter_std": 0.07780593633651733, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9728760123252869, "reward_repeat_soft_std": 0.0253890473395586, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.6060123443603516, "reward_total_composite_std": 0.022973021492362022} {"timestamp_utc": "2026-04-13T11:20:57Z", "mode": "train", "global_step": 1630, "epoch": 0.16373681567051732, "loss": -0.061, "grad_norm": 17.314697265625, "learning_rate": 5.063636363636364e-06, "num_tokens": 2890344.0, "completions/mean_length": 31.875, "completions/min_length": 23.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.875, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.8703351616859436, "rewards/meter/std": 0.1738702803850174, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.7987500429153442, "rewards/judge_quality/std": 0.22465451061725616, "rewards/total_composite/mean": 0.7996700406074524, "rewards/total_composite/std": 0.1645873337984085, "reward": 0.7996700406074524, "reward_std": 0.16458731889724731, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12682412564754486, "sampling/sampling_logp_difference/max": 1.3500020503997803, "sampling/importance_sampling_ratio/min": 0.25923970341682434, "sampling/importance_sampling_ratio/mean": 1.004669189453125, "sampling/importance_sampling_ratio/max": 1.9404537677764893, "entropy": 0.7230124175548553, "clip_ratio/low_mean": 0.035490778274834156, "clip_ratio/low_min": 0.035490778274834156, "clip_ratio/high_mean": 0.0786974485963583, "clip_ratio/high_max": 0.0786974485963583, "clip_ratio/region_mean": 0.11418822687119246, "reward_total_mean": 0.7996700406074524, "reward_meter_mean": 0.8703351616859436, "reward_meter_std": 0.1738702803850174, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.7987500429153442, "reward_judge_quality_std": 0.22465451061725616, "reward_total_composite_mean": 0.7996700406074524, "reward_total_composite_std": 0.1645873337984085} {"timestamp_utc": "2026-04-13T11:21:03Z", "mode": "train", "global_step": 1631, "epoch": 0.16383726770467102, "loss": 0.0586, "grad_norm": 12.814247131347656, "learning_rate": 5.060606060606061e-06, "num_tokens": 2892114.0, "completions/mean_length": 39.25, "completions/min_length": 34.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.25, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.7549862861633301, "rewards/meter/std": 0.29110273718833923, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9077358245849609, "rewards/repeat_soft/std": 0.044362038373947144, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.5813066959381104, "rewards/total_composite/std": 0.153243288397789, "reward": 0.5813066959381104, "reward_std": 0.1532432734966278, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09214208275079727, "sampling/sampling_logp_difference/max": 1.1270325183868408, "sampling/importance_sampling_ratio/min": 0.3239932656288147, "sampling/importance_sampling_ratio/mean": 1.006922721862793, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5208986103534698, "clip_ratio/low_mean": 0.02512002200819552, "clip_ratio/low_min": 0.02512002200819552, "clip_ratio/high_mean": 0.04087544255889952, "clip_ratio/high_max": 0.04087544255889952, "clip_ratio/region_mean": 0.06599546456709504, "reward_total_mean": 0.5813066959381104, "reward_meter_mean": 0.7549862861633301, "reward_meter_std": 0.29110273718833923, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9077358245849609, "reward_repeat_soft_std": 0.044362038373947144, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.5813066959381104, "reward_total_composite_std": 0.153243288397789} {"timestamp_utc": "2026-04-13T11:21:09Z", "mode": "train", "global_step": 1632, "epoch": 0.1639377197388247, "loss": 0.0718, "grad_norm": 12.439215660095215, "learning_rate": 5.057575757575758e-06, "num_tokens": 2893812.0, "completions/mean_length": 40.25, "completions/min_length": 34.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.25, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.6020994186401367, "rewards/meter/std": 0.3678845763206482, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9884793758392334, "rewards/repeat_soft/std": 0.014015505090355873, "rewards/judge_quality/mean": 0.5874999761581421, "rewards/judge_quality/std": 0.2043631225824356, "rewards/total_composite/mean": 0.5512148141860962, "rewards/total_composite/std": 0.1018567830324173, "reward": 0.5512148141860962, "reward_std": 0.1018567830324173, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.134989932179451, "sampling/sampling_logp_difference/max": 1.8583168983459473, "sampling/importance_sampling_ratio/min": 0.15593485534191132, "sampling/importance_sampling_ratio/mean": 1.0240408182144165, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8467823192477226, "clip_ratio/low_mean": 0.04137105867266655, "clip_ratio/low_min": 0.04137105867266655, "clip_ratio/high_mean": 0.08283164259046316, "clip_ratio/high_max": 0.08283164259046316, "clip_ratio/region_mean": 0.12420270126312971, "reward_total_mean": 0.5512148141860962, "reward_meter_mean": 0.6020994186401367, "reward_meter_std": 0.3678845763206482, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9884793758392334, "reward_repeat_soft_std": 0.014015505090355873, "reward_judge_quality_mean": 0.5874999761581421, "reward_judge_quality_std": 0.2043631225824356, "reward_total_composite_mean": 0.5512148141860962, "reward_total_composite_std": 0.1018567830324173} {"timestamp_utc": "2026-04-13T11:21:15Z", "mode": "train", "global_step": 1633, "epoch": 0.16403817177297841, "loss": 0.0113, "grad_norm": 10.988968849182129, "learning_rate": 5.054545454545455e-06, "num_tokens": 2895417.0, "completions/mean_length": 53.625, "completions/min_length": 40.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.625, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9202272891998291, "rewards/meter/std": 0.08080322295427322, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9281179904937744, "rewards/repeat_soft/std": 0.07783205807209015, "rewards/judge_quality/mean": 0.3812499940395355, "rewards/judge_quality/std": 0.08166787773370743, "rewards/total_composite/mean": 0.5684842467308044, "rewards/total_composite/std": 0.06606169790029526, "reward": 0.5684842467308044, "reward_std": 0.06606169790029526, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11435886472463608, "sampling/sampling_logp_difference/max": 1.7910586595535278, "sampling/importance_sampling_ratio/min": 0.16678351163864136, "sampling/importance_sampling_ratio/mean": 1.0170116424560547, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6273412480950356, "clip_ratio/low_mean": 0.019968220964074135, "clip_ratio/low_min": 0.019968220964074135, "clip_ratio/high_mean": 0.07719047367572784, "clip_ratio/high_max": 0.07719047367572784, "clip_ratio/region_mean": 0.09715869463980198, "reward_total_mean": 0.5684842467308044, "reward_meter_mean": 0.9202272891998291, "reward_meter_std": 0.08080322295427322, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9281179904937744, "reward_repeat_soft_std": 0.07783205807209015, "reward_judge_quality_mean": 0.3812499940395355, "reward_judge_quality_std": 0.08166787773370743, "reward_total_composite_mean": 0.5684842467308044, "reward_total_composite_std": 0.06606169790029526} {"timestamp_utc": "2026-04-13T11:21:28Z", "mode": "train", "global_step": 1634, "epoch": 0.1641386238071321, "loss": -0.1002, "grad_norm": 3.135519504547119, "learning_rate": 5.051515151515151e-06, "num_tokens": 2896975.0, "completions/mean_length": 112.75, "completions/min_length": 36.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 55.71428680419922, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.7123268842697144, "rewards/meter/std": 0.4074714779853821, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9069777727127075, "rewards/repeat_soft/std": 0.09436126053333282, "rewards/judge_quality/mean": 0.5612499713897705, "rewards/judge_quality/std": 0.3223324716091156, "rewards/total_composite/mean": 0.598398745059967, "rewards/total_composite/std": 0.3048110604286194, "reward": 0.598398745059967, "reward_std": 0.3048110604286194, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.142530158162117, "sampling/sampling_logp_difference/max": 2.1631574630737305, "sampling/importance_sampling_ratio/min": 0.11496156454086304, "sampling/importance_sampling_ratio/mean": 1.003870964050293, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7669676654040813, "clip_ratio/low_mean": 0.03367326036095619, "clip_ratio/low_min": 0.03367326036095619, "clip_ratio/high_mean": 0.06216681748628616, "clip_ratio/high_max": 0.06216681748628616, "clip_ratio/region_mean": 0.09584007784724236, "reward_total_mean": 0.598398745059967, "reward_meter_mean": 0.7123268842697144, "reward_meter_std": 0.4074714779853821, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9069777727127075, "reward_repeat_soft_std": 0.09436126053333282, "reward_judge_quality_mean": 0.5612499713897705, "reward_judge_quality_std": 0.3223324716091156, "reward_total_composite_mean": 0.598398745059967, "reward_total_composite_std": 0.3048110604286194} {"timestamp_utc": "2026-04-13T11:21:37Z", "mode": "train", "global_step": 1635, "epoch": 0.16423907584128578, "loss": 0.0696, "grad_norm": 13.71398639678955, "learning_rate": 5.048484848484849e-06, "num_tokens": 2898864.0, "completions/mean_length": 58.125, "completions/min_length": 50.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.125, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.7918387055397034, "rewards/meter/std": 0.37128695845603943, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9755090475082397, "rewards/repeat_soft/std": 0.019028011709451675, "rewards/judge_quality/mean": 0.7362500429153442, "rewards/judge_quality/std": 0.25376805663108826, "rewards/total_composite/mean": 0.7015790939331055, "rewards/total_composite/std": 0.21124167740345, "reward": 0.7015790939331055, "reward_std": 0.21124167740345, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1161414384841919, "sampling/sampling_logp_difference/max": 3.696103096008301, "sampling/importance_sampling_ratio/min": 0.024820059537887573, "sampling/importance_sampling_ratio/mean": 1.0041608810424805, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6581574194133282, "clip_ratio/low_mean": 0.07275632116943598, "clip_ratio/low_min": 0.07275632116943598, "clip_ratio/high_mean": 0.03863636404275894, "clip_ratio/high_max": 0.03863636404275894, "clip_ratio/region_mean": 0.11139268521219492, "reward_total_mean": 0.7015790939331055, "reward_meter_mean": 0.7918387055397034, "reward_meter_std": 0.37128695845603943, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9755090475082397, "reward_repeat_soft_std": 0.019028011709451675, "reward_judge_quality_mean": 0.7362500429153442, "reward_judge_quality_std": 0.25376805663108826, "reward_total_composite_mean": 0.7015790939331055, "reward_total_composite_std": 0.21124167740345} {"timestamp_utc": "2026-04-13T11:21:44Z", "mode": "train", "global_step": 1636, "epoch": 0.16433952787543948, "loss": -0.0376, "grad_norm": 17.088523864746094, "learning_rate": 5.045454545454546e-06, "num_tokens": 2900494.0, "completions/mean_length": 51.75, "completions/min_length": 42.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.75, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.6968623399734497, "rewards/meter/std": 0.33356785774230957, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8890321254730225, "rewards/repeat_soft/std": 0.058889240026474, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.5805325508117676, "rewards/total_composite/std": 0.16673095524311066, "reward": 0.5805325508117676, "reward_std": 0.16673094034194946, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1255987584590912, "sampling/sampling_logp_difference/max": 1.909731149673462, "sampling/importance_sampling_ratio/min": 0.1481201946735382, "sampling/importance_sampling_ratio/mean": 0.9984428286552429, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8163897320628166, "clip_ratio/low_mean": 0.046758043579757214, "clip_ratio/low_min": 0.046758043579757214, "clip_ratio/high_mean": 0.055573392659425735, "clip_ratio/high_max": 0.055573392659425735, "clip_ratio/region_mean": 0.10233143623918295, "reward_total_mean": 0.5805325508117676, "reward_meter_mean": 0.6968623399734497, "reward_meter_std": 0.33356785774230957, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8890321254730225, "reward_repeat_soft_std": 0.058889240026474, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.5805325508117676, "reward_total_composite_std": 0.16673095524311066} {"timestamp_utc": "2026-04-13T11:21:53Z", "mode": "train", "global_step": 1637, "epoch": 0.16443997990959316, "loss": 0.668, "grad_norm": 8.759028434753418, "learning_rate": 5.042424242424243e-06, "num_tokens": 2902433.0, "completions/mean_length": 90.375, "completions/min_length": 53.0, "completions/max_length": 284.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.375, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 284.0, "rewards/meter/mean": 0.6503992080688477, "rewards/meter/std": 0.4289567172527313, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9526631832122803, "rewards/repeat_soft/std": 0.038023222237825394, "rewards/judge_quality/mean": 0.4449999928474426, "rewards/judge_quality/std": 0.21876277029514313, "rewards/total_composite/mean": 0.5055826306343079, "rewards/total_composite/std": 0.24735192954540253, "reward": 0.5055826306343079, "reward_std": 0.24735191464424133, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12658533453941345, "sampling/sampling_logp_difference/max": 1.3045964241027832, "sampling/importance_sampling_ratio/min": 0.27128201723098755, "sampling/importance_sampling_ratio/mean": 1.0155060291290283, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.804808035492897, "clip_ratio/low_mean": 0.02805097633972764, "clip_ratio/low_min": 0.02805097633972764, "clip_ratio/high_mean": 0.07687525823712349, "clip_ratio/high_max": 0.07687525823712349, "clip_ratio/region_mean": 0.10492623457685113, "reward_total_mean": 0.5055826306343079, "reward_meter_mean": 0.6503992080688477, "reward_meter_std": 0.4289567172527313, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9526631832122803, "reward_repeat_soft_std": 0.038023222237825394, "reward_judge_quality_mean": 0.4449999928474426, "reward_judge_quality_std": 0.21876277029514313, "reward_total_composite_mean": 0.5055826306343079, "reward_total_composite_std": 0.24735192954540253} {"timestamp_utc": "2026-04-13T11:22:04Z", "mode": "train", "global_step": 1638, "epoch": 0.16454043194374687, "loss": -0.2145, "grad_norm": 2.6967246532440186, "learning_rate": 5.0393939393939395e-06, "num_tokens": 2904947.0, "completions/mean_length": 174.25, "completions/min_length": 109.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 126.00000762939453, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 160.0, "rewards/meter/mean": 0.9586606621742249, "rewards/meter/std": 0.060762591660022736, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.18516401946544647, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.7616796493530273, "rewards/repeat_soft/std": 0.16934245824813843, "rewards/judge_quality/mean": 0.3187499940395355, "rewards/judge_quality/std": 0.14961259067058563, "rewards/total_composite/mean": 0.44481372833251953, "rewards/total_composite/std": 0.2030736356973648, "reward": 0.44481372833251953, "reward_std": 0.2030736207962036, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10499298572540283, "sampling/sampling_logp_difference/max": 2.0107712745666504, "sampling/importance_sampling_ratio/min": 0.13388536870479584, "sampling/importance_sampling_ratio/mean": 1.0194429159164429, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7150713615119457, "clip_ratio/low_mean": 0.017359061632305384, "clip_ratio/low_min": 0.017359061632305384, "clip_ratio/high_mean": 0.07144996337592602, "clip_ratio/high_max": 0.07144996337592602, "clip_ratio/region_mean": 0.0888090250082314, "reward_total_mean": 0.44481372833251953, "reward_meter_mean": 0.9586606621742249, "reward_meter_std": 0.060762591660022736, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.18516401946544647, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.7616796493530273, "reward_repeat_soft_std": 0.16934245824813843, "reward_judge_quality_mean": 0.3187499940395355, "reward_judge_quality_std": 0.14961259067058563, "reward_total_composite_mean": 0.44481372833251953, "reward_total_composite_std": 0.2030736356973648} {"timestamp_utc": "2026-04-13T11:22:15Z", "mode": "train", "global_step": 1639, "epoch": 0.16464088397790055, "loss": -0.1139, "grad_norm": 2.492732048034668, "learning_rate": 5.036363636363637e-06, "num_tokens": 2906677.0, "completions/mean_length": 118.25, "completions/min_length": 50.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 62.000003814697266, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.8640415072441101, "rewards/meter/std": 0.32087600231170654, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9889659285545349, "rewards/repeat_soft/std": 0.020190119743347168, "rewards/judge_quality/mean": 0.6862499713897705, "rewards/judge_quality/std": 0.34221702814102173, "rewards/total_composite/mean": 0.7352422475814819, "rewards/total_composite/std": 0.329959511756897, "reward": 0.7352422475814819, "reward_std": 0.329959511756897, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14623674750328064, "sampling/sampling_logp_difference/max": 1.821061611175537, "sampling/importance_sampling_ratio/min": 0.1618538349866867, "sampling/importance_sampling_ratio/mean": 1.0083389282226562, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7862942218780518, "clip_ratio/low_mean": 0.02758458722382784, "clip_ratio/low_min": 0.02758458722382784, "clip_ratio/high_mean": 0.09788673743605614, "clip_ratio/high_max": 0.09788673743605614, "clip_ratio/region_mean": 0.12547132465988398, "reward_total_mean": 0.7352422475814819, "reward_meter_mean": 0.8640415072441101, "reward_meter_std": 0.32087600231170654, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9889659285545349, "reward_repeat_soft_std": 0.020190119743347168, "reward_judge_quality_mean": 0.6862499713897705, "reward_judge_quality_std": 0.34221702814102173, "reward_total_composite_mean": 0.7352422475814819, "reward_total_composite_std": 0.329959511756897} {"timestamp_utc": "2026-04-13T11:22:21Z", "mode": "train", "global_step": 1640, "epoch": 0.16474133601205423, "loss": 0.0875, "grad_norm": 10.618043899536133, "learning_rate": 5.033333333333333e-06, "num_tokens": 2908302.0, "completions/mean_length": 36.125, "completions/min_length": 30.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.125, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.7051137089729309, "rewards/meter/std": 0.36694246530532837, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9104615449905396, "rewards/repeat_soft/std": 0.08185256272554398, "rewards/judge_quality/mean": 0.5562499761581421, "rewards/judge_quality/std": 0.2249404937028885, "rewards/total_composite/mean": 0.5875173807144165, "rewards/total_composite/std": 0.16259489953517914, "reward": 0.5875173807144165, "reward_std": 0.16259488463401794, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08619284629821777, "sampling/sampling_logp_difference/max": 1.7844700813293457, "sampling/importance_sampling_ratio/min": 0.16788600385189056, "sampling/importance_sampling_ratio/mean": 1.0153688192367554, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.47819822654128075, "clip_ratio/low_mean": 0.023364828201010823, "clip_ratio/low_min": 0.023364828201010823, "clip_ratio/high_mean": 0.0380457378923893, "clip_ratio/high_max": 0.0380457378923893, "clip_ratio/region_mean": 0.06141056609340012, "reward_total_mean": 0.5875173807144165, "reward_meter_mean": 0.7051137089729309, "reward_meter_std": 0.36694246530532837, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9104615449905396, "reward_repeat_soft_std": 0.08185256272554398, "reward_judge_quality_mean": 0.5562499761581421, "reward_judge_quality_std": 0.2249404937028885, "reward_total_composite_mean": 0.5875173807144165, "reward_total_composite_std": 0.16259489953517914} {"timestamp_utc": "2026-04-13T11:22:29Z", "mode": "train", "global_step": 1641, "epoch": 0.16484178804620794, "loss": -0.0252, "grad_norm": 8.12309741973877, "learning_rate": 5.030303030303031e-06, "num_tokens": 2910322.0, "completions/mean_length": 83.5, "completions/min_length": 60.0, "completions/max_length": 111.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 83.5, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.9841760396957397, "rewards/meter/std": 0.008196291513741016, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8510419130325317, "rewards/repeat_soft/std": 0.07041285187005997, "rewards/judge_quality/mean": 0.5237500071525574, "rewards/judge_quality/std": 0.19078317284584045, "rewards/total_composite/mean": 0.6627941131591797, "rewards/total_composite/std": 0.1269497275352478, "reward": 0.6627941131591797, "reward_std": 0.1269497275352478, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10205010324716568, "sampling/sampling_logp_difference/max": 1.4837443828582764, "sampling/importance_sampling_ratio/min": 0.22678692638874054, "sampling/importance_sampling_ratio/mean": 1.0135223865509033, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.634688638150692, "clip_ratio/low_mean": 0.05080424412153661, "clip_ratio/low_min": 0.05080424412153661, "clip_ratio/high_mean": 0.025846945121884346, "clip_ratio/high_max": 0.025846945121884346, "clip_ratio/region_mean": 0.07665118924342096, "reward_total_mean": 0.6627941131591797, "reward_meter_mean": 0.9841760396957397, "reward_meter_std": 0.008196291513741016, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8510419130325317, "reward_repeat_soft_std": 0.07041285187005997, "reward_judge_quality_mean": 0.5237500071525574, "reward_judge_quality_std": 0.19078317284584045, "reward_total_composite_mean": 0.6627941131591797, "reward_total_composite_std": 0.1269497275352478} {"timestamp_utc": "2026-04-13T11:22:41Z", "mode": "train", "global_step": 1642, "epoch": 0.16494224008036162, "loss": -0.202, "grad_norm": 2.724552869796753, "learning_rate": 5.027272727272728e-06, "num_tokens": 2912716.0, "completions/mean_length": 162.25, "completions/min_length": 100.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 112.28572082519531, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.6643840074539185, "rewards/meter/std": 0.3807678520679474, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9060827493667603, "rewards/repeat_soft/std": 0.06389499455690384, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.13593590259552002, "rewards/total_composite/mean": 0.46449315547943115, "rewards/total_composite/std": 0.20578594505786896, "reward": 0.46449315547943115, "reward_std": 0.20578591525554657, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10945215076208115, "sampling/sampling_logp_difference/max": 2.316504955291748, "sampling/importance_sampling_ratio/min": 0.09861765801906586, "sampling/importance_sampling_ratio/mean": 1.0101243257522583, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6034212112426758, "clip_ratio/low_mean": 0.03504893695935607, "clip_ratio/low_min": 0.03504893695935607, "clip_ratio/high_mean": 0.049963342025876045, "clip_ratio/high_max": 0.049963342025876045, "clip_ratio/region_mean": 0.08501227898523211, "reward_total_mean": 0.46449315547943115, "reward_meter_mean": 0.6643840074539185, "reward_meter_std": 0.3807678520679474, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9060827493667603, "reward_repeat_soft_std": 0.06389499455690384, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.13593590259552002, "reward_total_composite_mean": 0.46449315547943115, "reward_total_composite_std": 0.20578594505786896} {"timestamp_utc": "2026-04-13T11:22:53Z", "mode": "train", "global_step": 1643, "epoch": 0.16504269211451533, "loss": -0.0798, "grad_norm": 1.5759207010269165, "learning_rate": 5.024242424242425e-06, "num_tokens": 2914122.0, "completions/mean_length": 216.75, "completions/min_length": 33.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 39.60000228881836, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.8150025606155396, "rewards/meter/std": 0.33757075667381287, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.9947206974029541, "rewards/repeat_soft/std": 0.00519803911447525, "rewards/judge_quality/mean": 0.29750001430511475, "rewards/judge_quality/std": 0.24188250303268433, "rewards/total_composite/mean": 0.3926314413547516, "rewards/total_composite/std": 0.3347933888435364, "reward": 0.3926314413547516, "reward_std": 0.334793359041214, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18363401293754578, "sampling/sampling_logp_difference/max": 1.8588404655456543, "sampling/importance_sampling_ratio/min": 0.1558532416820526, "sampling/importance_sampling_ratio/mean": 1.0017668008804321, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6053504720330238, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.1028136657550931, "clip_ratio/high_max": 0.1028136657550931, "clip_ratio/region_mean": 0.1028136657550931, "reward_total_mean": 0.3926314413547516, "reward_meter_mean": 0.8150025606155396, "reward_meter_std": 0.33757075667381287, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.9947206974029541, "reward_repeat_soft_std": 0.00519803911447525, "reward_judge_quality_mean": 0.29750001430511475, "reward_judge_quality_std": 0.24188250303268433, "reward_total_composite_mean": 0.3926314413547516, "reward_total_composite_std": 0.3347933888435364} {"timestamp_utc": "2026-04-13T11:22:59Z", "mode": "train", "global_step": 1644, "epoch": 0.165143144148669, "loss": 0.0057, "grad_norm": 10.083914756774902, "learning_rate": 5.021212121212121e-06, "num_tokens": 2915782.0, "completions/mean_length": 47.5, "completions/min_length": 41.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.5, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.7238696813583374, "rewards/meter/std": 0.3476002812385559, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9268432855606079, "rewards/repeat_soft/std": 0.035069454461336136, "rewards/judge_quality/mean": 0.5862500071525574, "rewards/judge_quality/std": 0.22984081506729126, "rewards/total_composite/mean": 0.6061728596687317, "rewards/total_composite/std": 0.14002163708209991, "reward": 0.6061728596687317, "reward_std": 0.14002162218093872, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09343069046735764, "sampling/sampling_logp_difference/max": 1.1515750885009766, "sampling/importance_sampling_ratio/min": 0.31613844633102417, "sampling/importance_sampling_ratio/mean": 1.0022430419921875, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5562189929187298, "clip_ratio/low_mean": 0.03513571806252003, "clip_ratio/low_min": 0.03513571806252003, "clip_ratio/high_mean": 0.04867568798363209, "clip_ratio/high_max": 0.04867568798363209, "clip_ratio/region_mean": 0.08381140604615211, "reward_total_mean": 0.6061728596687317, "reward_meter_mean": 0.7238696813583374, "reward_meter_std": 0.3476002812385559, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9268432855606079, "reward_repeat_soft_std": 0.035069454461336136, "reward_judge_quality_mean": 0.5862500071525574, "reward_judge_quality_std": 0.22984081506729126, "reward_total_composite_mean": 0.6061728596687317, "reward_total_composite_std": 0.14002163708209991} {"timestamp_utc": "2026-04-13T11:23:06Z", "mode": "train", "global_step": 1645, "epoch": 0.1652435961828227, "loss": -0.0361, "grad_norm": 17.102134704589844, "learning_rate": 5.0181818181818186e-06, "num_tokens": 2917231.0, "completions/mean_length": 27.125, "completions/min_length": 23.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.125, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.7270609140396118, "rewards/meter/std": 0.428354412317276, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9311273097991943, "rewards/repeat_soft/std": 0.07535543292760849, "rewards/judge_quality/mean": 0.5487500429153442, "rewards/judge_quality/std": 0.22937415540218353, "rewards/total_composite/mean": 0.5775827169418335, "rewards/total_composite/std": 0.17093253135681152, "reward": 0.5775827169418335, "reward_std": 0.17093253135681152, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1762944906949997, "sampling/sampling_logp_difference/max": 1.8742389678955078, "sampling/importance_sampling_ratio/min": 0.15347172319889069, "sampling/importance_sampling_ratio/mean": 1.0031790733337402, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.177955947816372, "clip_ratio/low_mean": 0.015416666865348816, "clip_ratio/low_min": 0.015416666865348816, "clip_ratio/high_mean": 0.129683512263, "clip_ratio/high_max": 0.129683512263, "clip_ratio/region_mean": 0.14510017912834883, "reward_total_mean": 0.5775827169418335, "reward_meter_mean": 0.7270609140396118, "reward_meter_std": 0.428354412317276, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9311273097991943, "reward_repeat_soft_std": 0.07535543292760849, "reward_judge_quality_mean": 0.5487500429153442, "reward_judge_quality_std": 0.22937415540218353, "reward_total_composite_mean": 0.5775827169418335, "reward_total_composite_std": 0.17093253135681152} {"timestamp_utc": "2026-04-13T11:23:13Z", "mode": "train", "global_step": 1646, "epoch": 0.1653440482169764, "loss": 0.0579, "grad_norm": 13.652124404907227, "learning_rate": 5.015151515151515e-06, "num_tokens": 2918812.0, "completions/mean_length": 40.625, "completions/min_length": 34.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.625, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.7669195532798767, "rewards/meter/std": 0.32380133867263794, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8907057046890259, "rewards/repeat_soft/std": 0.1519497185945511, "rewards/judge_quality/mean": 0.3812499940395355, "rewards/judge_quality/std": 0.08166787773370743, "rewards/total_composite/mean": 0.5211167931556702, "rewards/total_composite/std": 0.09286314249038696, "reward": 0.5211167931556702, "reward_std": 0.09286312758922577, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11913299560546875, "sampling/sampling_logp_difference/max": 1.107813835144043, "sampling/importance_sampling_ratio/min": 0.33028021454811096, "sampling/importance_sampling_ratio/mean": 1.0003385543823242, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.713285006582737, "clip_ratio/low_mean": 0.04548611235804856, "clip_ratio/low_min": 0.04548611235804856, "clip_ratio/high_mean": 0.07731873355805874, "clip_ratio/high_max": 0.07731873355805874, "clip_ratio/region_mean": 0.1228048459161073, "reward_total_mean": 0.5211167931556702, "reward_meter_mean": 0.7669195532798767, "reward_meter_std": 0.32380133867263794, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8907057046890259, "reward_repeat_soft_std": 0.1519497185945511, "reward_judge_quality_mean": 0.3812499940395355, "reward_judge_quality_std": 0.08166787773370743, "reward_total_composite_mean": 0.5211167931556702, "reward_total_composite_std": 0.09286314249038696} {"timestamp_utc": "2026-04-13T11:23:20Z", "mode": "train", "global_step": 1647, "epoch": 0.16544450025113008, "loss": 0.0052, "grad_norm": 11.083194732666016, "learning_rate": 5.012121212121212e-06, "num_tokens": 2920479.0, "completions/mean_length": 60.375, "completions/min_length": 53.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.375, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.6427329778671265, "rewards/meter/std": 0.25448077917099, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9895933866500854, "rewards/repeat_soft/std": 0.014044932089745998, "rewards/judge_quality/mean": 0.5900000333786011, "rewards/judge_quality/std": 0.27994897961616516, "rewards/total_composite/mean": 0.5815964937210083, "rewards/total_composite/std": 0.1397792249917984, "reward": 0.5815964937210083, "reward_std": 0.1397792249917984, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1341121792793274, "sampling/sampling_logp_difference/max": 1.6723310947418213, "sampling/importance_sampling_ratio/min": 0.18780875205993652, "sampling/importance_sampling_ratio/mean": 1.0191738605499268, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8402115479111671, "clip_ratio/low_mean": 0.05557193420827389, "clip_ratio/low_min": 0.05557193420827389, "clip_ratio/high_mean": 0.07705580070614815, "clip_ratio/high_max": 0.07705580070614815, "clip_ratio/region_mean": 0.13262773491442204, "reward_total_mean": 0.5815964937210083, "reward_meter_mean": 0.6427329778671265, "reward_meter_std": 0.25448077917099, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9895933866500854, "reward_repeat_soft_std": 0.014044932089745998, "reward_judge_quality_mean": 0.5900000333786011, "reward_judge_quality_std": 0.27994897961616516, "reward_total_composite_mean": 0.5815964937210083, "reward_total_composite_std": 0.1397792249917984} {"timestamp_utc": "2026-04-13T11:23:27Z", "mode": "train", "global_step": 1648, "epoch": 0.16554495228528376, "loss": -0.0047, "grad_norm": 6.380144119262695, "learning_rate": 5.009090909090909e-06, "num_tokens": 2922243.0, "completions/mean_length": 53.5, "completions/min_length": 46.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.5, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9764459133148193, "rewards/meter/std": 0.019814912229776382, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8258414268493652, "rewards/repeat_soft/std": 0.12010334432125092, "rewards/judge_quality/mean": 0.46875, "rewards/judge_quality/std": 0.19334924221038818, "rewards/total_composite/mean": 0.6210887432098389, "rewards/total_composite/std": 0.1282896101474762, "reward": 0.6210887432098389, "reward_std": 0.1282896101474762, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11061302572488785, "sampling/sampling_logp_difference/max": 1.0346717834472656, "sampling/importance_sampling_ratio/min": 0.3553430140018463, "sampling/importance_sampling_ratio/mean": 1.0079593658447266, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7249104119837284, "clip_ratio/low_mean": 0.061141025042161345, "clip_ratio/low_min": 0.061141025042161345, "clip_ratio/high_mean": 0.040674603544175625, "clip_ratio/high_max": 0.040674603544175625, "clip_ratio/region_mean": 0.10181562858633697, "reward_total_mean": 0.6210887432098389, "reward_meter_mean": 0.9764459133148193, "reward_meter_std": 0.019814912229776382, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8258414268493652, "reward_repeat_soft_std": 0.12010334432125092, "reward_judge_quality_mean": 0.46875, "reward_judge_quality_std": 0.19334924221038818, "reward_total_composite_mean": 0.6210887432098389, "reward_total_composite_std": 0.1282896101474762} {"timestamp_utc": "2026-04-13T11:23:33Z", "mode": "train", "global_step": 1649, "epoch": 0.16564540431943747, "loss": -0.001, "grad_norm": 8.52895450592041, "learning_rate": 5.006060606060607e-06, "num_tokens": 2923968.0, "completions/mean_length": 61.625, "completions/min_length": 54.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.625, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9864411354064941, "rewards/meter/std": 0.01547372154891491, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9763901829719543, "rewards/repeat_soft/std": 0.020429003983736038, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.11667262762784958, "rewards/total_composite/mean": 0.6421948671340942, "rewards/total_composite/std": 0.07390771061182022, "reward": 0.6421948671340942, "reward_std": 0.07390771806240082, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1188473328948021, "sampling/sampling_logp_difference/max": 1.487046718597412, "sampling/importance_sampling_ratio/min": 0.22603923082351685, "sampling/importance_sampling_ratio/mean": 1.012412428855896, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9211339130997658, "clip_ratio/low_mean": 0.08232017001137137, "clip_ratio/low_min": 0.08232017001137137, "clip_ratio/high_mean": 0.015625, "clip_ratio/high_max": 0.015625, "clip_ratio/region_mean": 0.09794517001137137, "reward_total_mean": 0.6421948671340942, "reward_meter_mean": 0.9864411354064941, "reward_meter_std": 0.01547372154891491, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9763901829719543, "reward_repeat_soft_std": 0.020429003983736038, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.11667262762784958, "reward_total_composite_mean": 0.6421948671340942, "reward_total_composite_std": 0.07390771061182022} {"timestamp_utc": "2026-04-13T11:23:40Z", "mode": "train", "global_step": 1650, "epoch": 0.16574585635359115, "loss": -0.0375, "grad_norm": 9.283758163452148, "learning_rate": 5.003030303030303e-06, "num_tokens": 2925857.0, "completions/mean_length": 54.125, "completions/min_length": 51.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.125, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9835355877876282, "rewards/meter/std": 0.01232390571385622, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8640047907829285, "rewards/repeat_soft/std": 0.033385492861270905, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.6005144715309143, "rewards/total_composite/std": 0.010814270935952663, "reward": 0.6005144715309143, "reward_std": 0.010814269073307514, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10571616888046265, "sampling/sampling_logp_difference/max": 1.3834004402160645, "sampling/importance_sampling_ratio/min": 0.25072452425956726, "sampling/importance_sampling_ratio/mean": 1.0140377283096313, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5891980826854706, "clip_ratio/low_mean": 0.056812843307852745, "clip_ratio/low_min": 0.056812843307852745, "clip_ratio/high_mean": 0.04984511248767376, "clip_ratio/high_max": 0.04984511248767376, "clip_ratio/region_mean": 0.1066579557955265, "reward_total_mean": 0.6005144715309143, "reward_meter_mean": 0.9835355877876282, "reward_meter_std": 0.01232390571385622, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8640047907829285, "reward_repeat_soft_std": 0.033385492861270905, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.6005144715309143, "reward_total_composite_std": 0.010814270935952663} {"timestamp_utc": "2026-04-13T11:24:50Z", "mode": "eval", "global_step": 1650, "epoch": 0.16574585635359115, "eval_loss": NaN, "eval_runtime": 69.6645, "eval_samples_per_second": 1.148, "eval_steps_per_second": 0.144, "eval_num_tokens": 2925857.0, "eval_completions/mean_length": 115.75, "eval_completions/min_length": 37.9, "eval_completions/max_length": 347.1, "eval_completions/clipped_ratio": 0.0625, "eval_completions/mean_terminated_length": 88.40357284545898, "eval_completions/min_terminated_length": 37.9, "eval_completions/max_terminated_length": 160.2, "eval_rewards/meter/mean": 0.7606006145477295, "eval_rewards/meter/std": 0.30303533375263214, "eval_rewards/count_adherence/mean": 0.9537499904632568, "eval_rewards/count_adherence/std": 0.08891427256166935, "eval_rewards/hard_gate/mean": 0.925, "eval_rewards/hard_gate/std": 0.18771235942840575, "eval_rewards/repeat_soft/mean": 0.8713025391101837, "eval_rewards/repeat_soft/std": 0.10102997198700905, "eval_rewards/judge_quality/mean": 0.4048749953508377, "eval_rewards/judge_quality/std": 0.1862373486161232, "eval_rewards/total_composite/mean": 0.49997623860836027, "eval_rewards/total_composite/std": 0.18340689837932586, "eval_reward": 0.49997623860836027, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.05427191890776158, "eval_sampling/sampling_logp_difference/max": 0.9719480037689209, "eval_sampling/importance_sampling_ratio/min": 0.38455626368522644, "eval_sampling/importance_sampling_ratio/mean": 1.0114875197410584, "eval_sampling/importance_sampling_ratio/max": 1.4432477235794068, "eval_entropy": 0.612224942445755, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.49997623860836027, "eval_reward_meter_mean": 0.7606006145477295, "eval_reward_meter_std": 0.30303533375263214, "eval_reward_count_adherence_mean": 0.9537499904632568, "eval_reward_count_adherence_std": 0.08891427256166935, "eval_reward_hard_gate_mean": 0.925, "eval_reward_hard_gate_std": 0.18771235942840575, "eval_reward_repeat_soft_mean": 0.8713025391101837, "eval_reward_repeat_soft_std": 0.10102997198700905, "eval_reward_judge_quality_mean": 0.4048749953508377, "eval_reward_judge_quality_std": 0.1862373486161232, "eval_reward_total_composite_mean": 0.49997623860836027, "eval_reward_total_composite_std": 0.18340689837932586} {"timestamp_utc": "2026-04-13T11:25:00Z", "mode": "train", "global_step": 1651, "epoch": 0.16584630838774486, "loss": 0.0804, "grad_norm": 12.425683975219727, "learning_rate": 5e-06, "num_tokens": 2927346.0, "completions/mean_length": 44.125, "completions/min_length": 34.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.125, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9057676792144775, "rewards/meter/std": 0.121575728058815, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8857040405273438, "rewards/repeat_soft/std": 0.06436088681221008, "rewards/judge_quality/mean": 0.4312499761581421, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.5860784649848938, "rewards/total_composite/std": 0.027623461559414864, "reward": 0.5860784649848938, "reward_std": 0.027623457834124565, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09438827633857727, "sampling/sampling_logp_difference/max": 1.3431239128112793, "sampling/importance_sampling_ratio/min": 0.2610289752483368, "sampling/importance_sampling_ratio/mean": 1.0196468830108643, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6710145398974419, "clip_ratio/low_mean": 0.05590006709098816, "clip_ratio/low_min": 0.05590006709098816, "clip_ratio/high_mean": 0.05511593818664551, "clip_ratio/high_max": 0.05511593818664551, "clip_ratio/region_mean": 0.11101600527763367, "reward_total_mean": 0.5860784649848938, "reward_meter_mean": 0.9057676792144775, "reward_meter_std": 0.121575728058815, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8857040405273438, "reward_repeat_soft_std": 0.06436088681221008, "reward_judge_quality_mean": 0.4312499761581421, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.5860784649848938, "reward_total_composite_std": 0.027623461559414864} {"timestamp_utc": "2026-04-13T11:25:07Z", "mode": "train", "global_step": 1652, "epoch": 0.16594676042189854, "loss": 0.0635, "grad_norm": 7.243699550628662, "learning_rate": 4.996969696969698e-06, "num_tokens": 2929413.0, "completions/mean_length": 87.375, "completions/min_length": 75.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.375, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.9921964406967163, "rewards/meter/std": 0.0031764977611601353, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8434470891952515, "rewards/repeat_soft/std": 0.09848575294017792, "rewards/judge_quality/mean": 0.4024999737739563, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.5778021216392517, "rewards/total_composite/std": 0.04142162203788757, "reward": 0.5778021216392517, "reward_std": 0.041421618312597275, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07206101715564728, "sampling/sampling_logp_difference/max": 1.4006543159484863, "sampling/importance_sampling_ratio/min": 0.24643567204475403, "sampling/importance_sampling_ratio/mean": 1.0028376579284668, "sampling/importance_sampling_ratio/max": 1.7866092920303345, "entropy": 0.546353530138731, "clip_ratio/low_mean": 0.024673645850270987, "clip_ratio/low_min": 0.024673645850270987, "clip_ratio/high_mean": 0.06221241131424904, "clip_ratio/high_max": 0.06221241131424904, "clip_ratio/region_mean": 0.08688605716452003, "reward_total_mean": 0.5778021216392517, "reward_meter_mean": 0.9921964406967163, "reward_meter_std": 0.0031764977611601353, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8434470891952515, "reward_repeat_soft_std": 0.09848575294017792, "reward_judge_quality_mean": 0.4024999737739563, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.5778021216392517, "reward_total_composite_std": 0.04142162203788757} {"timestamp_utc": "2026-04-13T11:25:19Z", "mode": "train", "global_step": 1653, "epoch": 0.16604721245605222, "loss": -0.1949, "grad_norm": 1.912166953086853, "learning_rate": 4.993939393939394e-06, "num_tokens": 2931539.0, "completions/mean_length": 159.75, "completions/min_length": 90.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 109.42857360839844, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.9761732816696167, "rewards/meter/std": 0.022720949724316597, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8665366768836975, "rewards/repeat_soft/std": 0.044016994535923004, "rewards/judge_quality/mean": 0.4112499952316284, "rewards/judge_quality/std": 0.1797965168952942, "rewards/total_composite/mean": 0.5456495881080627, "rewards/total_composite/std": 0.23139263689517975, "reward": 0.5456495881080627, "reward_std": 0.23139263689517975, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12424559146165848, "sampling/sampling_logp_difference/max": 2.206782341003418, "sampling/importance_sampling_ratio/min": 0.11005418747663498, "sampling/importance_sampling_ratio/mean": 1.0153663158416748, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.656967006623745, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09762474708259106, "clip_ratio/high_max": 0.09762474708259106, "clip_ratio/region_mean": 0.09762474708259106, "reward_total_mean": 0.5456495881080627, "reward_meter_mean": 0.9761732816696167, "reward_meter_std": 0.022720949724316597, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8665366768836975, "reward_repeat_soft_std": 0.044016994535923004, "reward_judge_quality_mean": 0.4112499952316284, "reward_judge_quality_std": 0.1797965168952942, "reward_total_composite_mean": 0.5456495881080627, "reward_total_composite_std": 0.23139263689517975} {"timestamp_utc": "2026-04-13T11:25:31Z", "mode": "train", "global_step": 1654, "epoch": 0.16614766449020593, "loss": -0.1346, "grad_norm": 1.924111247062683, "learning_rate": 4.990909090909091e-06, "num_tokens": 2933194.0, "completions/mean_length": 110.875, "completions/min_length": 45.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 53.57143020629883, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.8242701292037964, "rewards/meter/std": 0.3262185752391815, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9393479824066162, "rewards/repeat_soft/std": 0.049659643322229385, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.13593590259552002, "rewards/total_composite/mean": 0.5063811540603638, "rewards/total_composite/std": 0.20991861820220947, "reward": 0.5063811540603638, "reward_std": 0.20991861820220947, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12239545583724976, "sampling/sampling_logp_difference/max": 3.181450843811035, "sampling/importance_sampling_ratio/min": 0.04152536392211914, "sampling/importance_sampling_ratio/mean": 0.9930113554000854, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5652026124298573, "clip_ratio/low_mean": 0.007936508394777775, "clip_ratio/low_min": 0.007936508394777775, "clip_ratio/high_mean": 0.09462250303477049, "clip_ratio/high_max": 0.09462250303477049, "clip_ratio/region_mean": 0.10255901142954826, "reward_total_mean": 0.5063811540603638, "reward_meter_mean": 0.8242701292037964, "reward_meter_std": 0.3262185752391815, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9393479824066162, "reward_repeat_soft_std": 0.049659643322229385, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.13593590259552002, "reward_total_composite_mean": 0.5063811540603638, "reward_total_composite_std": 0.20991861820220947} {"timestamp_utc": "2026-04-13T11:25:37Z", "mode": "train", "global_step": 1655, "epoch": 0.1662481165243596, "loss": 0.0539, "grad_norm": 17.1294002532959, "learning_rate": 4.987878787878789e-06, "num_tokens": 2934846.0, "completions/mean_length": 40.5, "completions/min_length": 35.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.5, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.7644696831703186, "rewards/meter/std": 0.2678174376487732, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9411360621452332, "rewards/repeat_soft/std": 0.0657808780670166, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.620587944984436, "rewards/total_composite/std": 0.16680742800235748, "reward": 0.620587944984436, "reward_std": 0.16680742800235748, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11558982729911804, "sampling/sampling_logp_difference/max": 2.733036518096924, "sampling/importance_sampling_ratio/min": 0.06502155214548111, "sampling/importance_sampling_ratio/mean": 1.0181759595870972, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4405171722173691, "clip_ratio/low_mean": 0.08305060723796487, "clip_ratio/low_min": 0.08305060723796487, "clip_ratio/high_mean": 0.02430124208331108, "clip_ratio/high_max": 0.02430124208331108, "clip_ratio/region_mean": 0.10735184932127595, "reward_total_mean": 0.620587944984436, "reward_meter_mean": 0.7644696831703186, "reward_meter_std": 0.2678174376487732, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9411360621452332, "reward_repeat_soft_std": 0.0657808780670166, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.620587944984436, "reward_total_composite_std": 0.16680742800235748} {"timestamp_utc": "2026-04-13T11:25:50Z", "mode": "train", "global_step": 1656, "epoch": 0.16634856855851332, "loss": -0.072, "grad_norm": 1.673622965812683, "learning_rate": 4.984848484848485e-06, "num_tokens": 2936323.0, "completions/mean_length": 80.625, "completions/min_length": 17.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 19.0, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 21.0, "rewards/meter/mean": 0.857839822769165, "rewards/meter/std": 0.32516783475875854, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9595891237258911, "rewards/repeat_soft/std": 0.006366936955600977, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.13593590259552002, "rewards/total_composite/mean": 0.5196996331214905, "rewards/total_composite/std": 0.21318161487579346, "reward": 0.5196996331214905, "reward_std": 0.21318161487579346, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10437224060297012, "sampling/sampling_logp_difference/max": 1.280099868774414, "sampling/importance_sampling_ratio/min": 0.2780095338821411, "sampling/importance_sampling_ratio/mean": 0.9901865720748901, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.49581820517778397, "clip_ratio/low_mean": 0.007352941203862429, "clip_ratio/low_min": 0.007352941203862429, "clip_ratio/high_mean": 0.10395781276747584, "clip_ratio/high_max": 0.10395781276747584, "clip_ratio/region_mean": 0.11131075397133827, "reward_total_mean": 0.5196996331214905, "reward_meter_mean": 0.857839822769165, "reward_meter_std": 0.32516783475875854, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9595891237258911, "reward_repeat_soft_std": 0.006366936955600977, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.13593590259552002, "reward_total_composite_mean": 0.5196996331214905, "reward_total_composite_std": 0.21318161487579346} {"timestamp_utc": "2026-04-13T11:25:57Z", "mode": "train", "global_step": 1657, "epoch": 0.166449020592667, "loss": 0.042, "grad_norm": 8.607499122619629, "learning_rate": 4.981818181818182e-06, "num_tokens": 2938306.0, "completions/mean_length": 80.875, "completions/min_length": 74.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.875, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.9017000794410706, "rewards/meter/std": 0.16966751217842102, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8983750343322754, "rewards/repeat_soft/std": 0.03722051531076431, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.619802713394165, "rewards/total_composite/std": 0.12518617510795593, "reward": 0.619802713394165, "reward_std": 0.12518619000911713, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12372054904699326, "sampling/sampling_logp_difference/max": 2.9575610160827637, "sampling/importance_sampling_ratio/min": 0.05194545537233353, "sampling/importance_sampling_ratio/mean": 1.0086854696273804, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5657336711883545, "clip_ratio/low_mean": 0.10132492426782846, "clip_ratio/low_min": 0.10132492426782846, "clip_ratio/high_mean": 0.01461038924753666, "clip_ratio/high_max": 0.01461038924753666, "clip_ratio/region_mean": 0.11593531351536512, "reward_total_mean": 0.619802713394165, "reward_meter_mean": 0.9017000794410706, "reward_meter_std": 0.16966751217842102, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8983750343322754, "reward_repeat_soft_std": 0.03722051531076431, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.619802713394165, "reward_total_composite_std": 0.12518617510795593} {"timestamp_utc": "2026-04-13T11:26:08Z", "mode": "train", "global_step": 1658, "epoch": 0.16654947262682068, "loss": -0.2136, "grad_norm": 1.5572788715362549, "learning_rate": 4.978787878787879e-06, "num_tokens": 2940588.0, "completions/mean_length": 168.25, "completions/min_length": 115.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 119.14286041259766, "completions/min_terminated_length": 115.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.8625526428222656, "rewards/meter/std": 0.3485822081565857, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8076668977737427, "rewards/repeat_soft/std": 0.11250165104866028, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.5128768682479858, "rewards/total_composite/std": 0.2076120376586914, "reward": 0.5128768682479858, "reward_std": 0.2076120227575302, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08549162745475769, "sampling/sampling_logp_difference/max": 5.35800313949585, "sampling/importance_sampling_ratio/min": 0.004710302222520113, "sampling/importance_sampling_ratio/mean": 1.001693606376648, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3563239946961403, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.06801584642380476, "clip_ratio/high_max": 0.06801584642380476, "clip_ratio/region_mean": 0.06801584642380476, "reward_total_mean": 0.5128768682479858, "reward_meter_mean": 0.8625526428222656, "reward_meter_std": 0.3485822081565857, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8076668977737427, "reward_repeat_soft_std": 0.11250165104866028, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.5128768682479858, "reward_total_composite_std": 0.2076120376586914} {"timestamp_utc": "2026-04-13T11:26:15Z", "mode": "train", "global_step": 1659, "epoch": 0.1666499246609744, "loss": -0.0036, "grad_norm": 6.4640045166015625, "learning_rate": 4.975757575757576e-06, "num_tokens": 2943088.0, "completions/mean_length": 118.5, "completions/min_length": 104.0, "completions/max_length": 130.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 118.5, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 130.0, "rewards/meter/mean": 0.890613317489624, "rewards/meter/std": 0.27192914485931396, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9004124999046326, "rewards/repeat_soft/std": 0.04215488210320473, "rewards/judge_quality/mean": 0.47749999165534973, "rewards/judge_quality/std": 0.2303258776664734, "rewards/total_composite/mean": 0.5966652035713196, "rewards/total_composite/std": 0.15086136758327484, "reward": 0.5966652035713196, "reward_std": 0.15086136758327484, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12144719064235687, "sampling/sampling_logp_difference/max": 3.8907980918884277, "sampling/importance_sampling_ratio/min": 0.020429035648703575, "sampling/importance_sampling_ratio/mean": 0.996963381767273, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5733271576464176, "clip_ratio/low_mean": 0.043103738222271204, "clip_ratio/low_min": 0.043103738222271204, "clip_ratio/high_mean": 0.06401834078133106, "clip_ratio/high_max": 0.06401834078133106, "clip_ratio/region_mean": 0.10712207900360227, "reward_total_mean": 0.5966652035713196, "reward_meter_mean": 0.890613317489624, "reward_meter_std": 0.27192914485931396, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9004124999046326, "reward_repeat_soft_std": 0.04215488210320473, "reward_judge_quality_mean": 0.47749999165534973, "reward_judge_quality_std": 0.2303258776664734, "reward_total_composite_mean": 0.5966652035713196, "reward_total_composite_std": 0.15086136758327484} {"timestamp_utc": "2026-04-13T11:26:22Z", "mode": "train", "global_step": 1660, "epoch": 0.16675037669512807, "loss": 0.0085, "grad_norm": 8.742388725280762, "learning_rate": 4.972727272727273e-06, "num_tokens": 2944760.0, "completions/mean_length": 44.0, "completions/min_length": 35.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.8351925611495972, "rewards/meter/std": 0.27675724029541016, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8899495005607605, "rewards/repeat_soft/std": 0.02758779190480709, "rewards/judge_quality/mean": 0.41749998927116394, "rewards/judge_quality/std": 0.06902380287647247, "rewards/total_composite/mean": 0.5577464699745178, "rewards/total_composite/std": 0.08479692786931992, "reward": 0.5577464699745178, "reward_std": 0.08479694277048111, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11134511977434158, "sampling/sampling_logp_difference/max": 1.9499166011810303, "sampling/importance_sampling_ratio/min": 0.14228594303131104, "sampling/importance_sampling_ratio/mean": 0.9990329742431641, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6182470545172691, "clip_ratio/low_mean": 0.027862208895385265, "clip_ratio/low_min": 0.027862208895385265, "clip_ratio/high_mean": 0.10116859432309866, "clip_ratio/high_max": 0.10116859432309866, "clip_ratio/region_mean": 0.12903080321848392, "reward_total_mean": 0.5577464699745178, "reward_meter_mean": 0.8351925611495972, "reward_meter_std": 0.27675724029541016, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8899495005607605, "reward_repeat_soft_std": 0.02758779190480709, "reward_judge_quality_mean": 0.41749998927116394, "reward_judge_quality_std": 0.06902380287647247, "reward_total_composite_mean": 0.5577464699745178, "reward_total_composite_std": 0.08479692786931992} {"timestamp_utc": "2026-04-13T11:26:28Z", "mode": "train", "global_step": 1661, "epoch": 0.16685082872928178, "loss": -0.0193, "grad_norm": 8.281079292297363, "learning_rate": 4.9696969696969696e-06, "num_tokens": 2946558.0, "completions/mean_length": 59.75, "completions/min_length": 49.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.75, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.9679084420204163, "rewards/meter/std": 0.031183168292045593, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9141935110092163, "rewards/repeat_soft/std": 0.06871075928211212, "rewards/judge_quality/mean": 0.5824999809265137, "rewards/judge_quality/std": 0.23260943591594696, "rewards/total_composite/mean": 0.7038965821266174, "rewards/total_composite/std": 0.15150462090969086, "reward": 0.7038965821266174, "reward_std": 0.15150462090969086, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09651017189025879, "sampling/sampling_logp_difference/max": 2.2903401851654053, "sampling/importance_sampling_ratio/min": 0.10123201459646225, "sampling/importance_sampling_ratio/mean": 1.0104434490203857, "sampling/importance_sampling_ratio/max": 1.9806965589523315, "entropy": 0.5305213257670403, "clip_ratio/low_mean": 0.038811630103737116, "clip_ratio/low_min": 0.038811630103737116, "clip_ratio/high_mean": 0.0311390720307827, "clip_ratio/high_max": 0.0311390720307827, "clip_ratio/region_mean": 0.06995070213451982, "reward_total_mean": 0.7038965821266174, "reward_meter_mean": 0.9679084420204163, "reward_meter_std": 0.031183168292045593, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9141935110092163, "reward_repeat_soft_std": 0.06871075928211212, "reward_judge_quality_mean": 0.5824999809265137, "reward_judge_quality_std": 0.23260943591594696, "reward_total_composite_mean": 0.7038965821266174, "reward_total_composite_std": 0.15150462090969086} {"timestamp_utc": "2026-04-13T11:26:36Z", "mode": "train", "global_step": 1662, "epoch": 0.16695128076343546, "loss": -0.0129, "grad_norm": 5.485751628875732, "learning_rate": 4.966666666666667e-06, "num_tokens": 2948814.0, "completions/mean_length": 117.0, "completions/min_length": 95.0, "completions/max_length": 139.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 117.0, "completions/min_terminated_length": 95.0, "completions/max_terminated_length": 139.0, "rewards/meter/mean": 0.8648841381072998, "rewards/meter/std": 0.16966873407363892, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6611753106117249, "rewards/repeat_soft/std": 0.055279362946748734, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.522525429725647, "rewards/total_composite/std": 0.06771218776702881, "reward": 0.522525429725647, "reward_std": 0.0677121952176094, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06389590352773666, "sampling/sampling_logp_difference/max": 1.5982273817062378, "sampling/importance_sampling_ratio/min": 0.20225471258163452, "sampling/importance_sampling_ratio/mean": 1.0021018981933594, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2970382459461689, "clip_ratio/low_mean": 0.03604708984494209, "clip_ratio/low_min": 0.03604708984494209, "clip_ratio/high_mean": 0.024507308844476938, "clip_ratio/high_max": 0.024507308844476938, "clip_ratio/region_mean": 0.06055439868941903, "reward_total_mean": 0.522525429725647, "reward_meter_mean": 0.8648841381072998, "reward_meter_std": 0.16966873407363892, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6611753106117249, "reward_repeat_soft_std": 0.055279362946748734, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.522525429725647, "reward_total_composite_std": 0.06771218776702881} {"timestamp_utc": "2026-04-13T11:26:48Z", "mode": "train", "global_step": 1663, "epoch": 0.16705173279758914, "loss": -0.1132, "grad_norm": 2.1583328247070312, "learning_rate": 4.963636363636364e-06, "num_tokens": 2950458.0, "completions/mean_length": 101.5, "completions/min_length": 36.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 42.85714340209961, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.872140645980835, "rewards/meter/std": 0.2246885895729065, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9181938767433167, "rewards/repeat_soft/std": 0.0227973610162735, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.23439893126487732, "rewards/total_composite/mean": 0.564523458480835, "rewards/total_composite/std": 0.2551872432231903, "reward": 0.564523458480835, "reward_std": 0.2551872432231903, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10454196482896805, "sampling/sampling_logp_difference/max": 2.0216989517211914, "sampling/importance_sampling_ratio/min": 0.1324302852153778, "sampling/importance_sampling_ratio/mean": 1.0126889944076538, "sampling/importance_sampling_ratio/max": 1.9438802003860474, "entropy": 0.6456299051642418, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.11845581140369177, "clip_ratio/high_max": 0.11845581140369177, "clip_ratio/region_mean": 0.11845581140369177, "reward_total_mean": 0.564523458480835, "reward_meter_mean": 0.872140645980835, "reward_meter_std": 0.2246885895729065, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9181938767433167, "reward_repeat_soft_std": 0.0227973610162735, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.23439893126487732, "reward_total_composite_mean": 0.564523458480835, "reward_total_composite_std": 0.2551872432231903} {"timestamp_utc": "2026-04-13T11:26:55Z", "mode": "train", "global_step": 1664, "epoch": 0.16715218483174285, "loss": 0.0608, "grad_norm": 18.309833526611328, "learning_rate": 4.9606060606060605e-06, "num_tokens": 2951779.0, "completions/mean_length": 27.125, "completions/min_length": 19.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.125, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.8375441431999207, "rewards/meter/std": 0.310146689414978, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9536747932434082, "rewards/repeat_soft/std": 0.012406822293996811, "rewards/judge_quality/mean": 0.39625000953674316, "rewards/judge_quality/std": 0.09085899591445923, "rewards/total_composite/mean": 0.5535157918930054, "rewards/total_composite/std": 0.08959093689918518, "reward": 0.5535157918930054, "reward_std": 0.08959092944860458, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12178554385900497, "sampling/sampling_logp_difference/max": 0.7920396327972412, "sampling/importance_sampling_ratio/min": 0.4529200494289398, "sampling/importance_sampling_ratio/mean": 1.027703046798706, "sampling/importance_sampling_ratio/max": 1.977829933166504, "entropy": 1.03718501329422, "clip_ratio/low_mean": 0.04992155823856592, "clip_ratio/low_min": 0.04992155823856592, "clip_ratio/high_mean": 0.07084077317267656, "clip_ratio/high_max": 0.07084077317267656, "clip_ratio/region_mean": 0.12076233141124249, "reward_total_mean": 0.5535157918930054, "reward_meter_mean": 0.8375441431999207, "reward_meter_std": 0.310146689414978, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9536747932434082, "reward_repeat_soft_std": 0.012406822293996811, "reward_judge_quality_mean": 0.39625000953674316, "reward_judge_quality_std": 0.09085899591445923, "reward_total_composite_mean": 0.5535157918930054, "reward_total_composite_std": 0.08959093689918518} {"timestamp_utc": "2026-04-13T11:27:01Z", "mode": "train", "global_step": 1665, "epoch": 0.16725263686589653, "loss": 0.0436, "grad_norm": 9.189441680908203, "learning_rate": 4.957575757575758e-06, "num_tokens": 2953601.0, "completions/mean_length": 56.75, "completions/min_length": 46.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.75, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.5839331150054932, "rewards/meter/std": 0.33002543449401855, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8933404088020325, "rewards/repeat_soft/std": 0.05125471204519272, "rewards/judge_quality/mean": 0.38874998688697815, "rewards/judge_quality/std": 0.08675704896450043, "rewards/total_composite/mean": 0.4816441833972931, "rewards/total_composite/std": 0.08596338331699371, "reward": 0.4816441833972931, "reward_std": 0.08596337586641312, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09927050769329071, "sampling/sampling_logp_difference/max": 2.0201175212860107, "sampling/importance_sampling_ratio/min": 0.13263988494873047, "sampling/importance_sampling_ratio/mean": 1.0076899528503418, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.44833118095993996, "clip_ratio/low_mean": 0.06354818027466536, "clip_ratio/low_min": 0.06354818027466536, "clip_ratio/high_mean": 0.03069634223356843, "clip_ratio/high_max": 0.03069634223356843, "clip_ratio/region_mean": 0.09424452250823379, "reward_total_mean": 0.4816441833972931, "reward_meter_mean": 0.5839331150054932, "reward_meter_std": 0.33002543449401855, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8933404088020325, "reward_repeat_soft_std": 0.05125471204519272, "reward_judge_quality_mean": 0.38874998688697815, "reward_judge_quality_std": 0.08675704896450043, "reward_total_composite_mean": 0.4816441833972931, "reward_total_composite_std": 0.08596338331699371} {"timestamp_utc": "2026-04-13T11:27:14Z", "mode": "train", "global_step": 1666, "epoch": 0.16735308890005024, "loss": -0.0541, "grad_norm": 4.4340715408325195, "learning_rate": 4.954545454545455e-06, "num_tokens": 2954941.0, "completions/mean_length": 83.5, "completions/min_length": 18.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 22.285715103149414, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.668758749961853, "rewards/meter/std": 0.44776076078414917, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9480493068695068, "rewards/repeat_soft/std": 0.03862285614013672, "rewards/judge_quality/mean": 0.26249998807907104, "rewards/judge_quality/std": 0.13562026619911194, "rewards/total_composite/mean": 0.3588586449623108, "rewards/total_composite/std": 0.23403485119342804, "reward": 0.3588586449623108, "reward_std": 0.23403485119342804, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15646137297153473, "sampling/sampling_logp_difference/max": 1.7519347667694092, "sampling/importance_sampling_ratio/min": 0.17343805730342865, "sampling/importance_sampling_ratio/mean": 1.018505573272705, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1319462731480598, "clip_ratio/low_mean": 0.024801588151603937, "clip_ratio/low_min": 0.024801588151603937, "clip_ratio/high_mean": 0.0837974464520812, "clip_ratio/high_max": 0.0837974464520812, "clip_ratio/region_mean": 0.10859903460368514, "reward_total_mean": 0.3588586449623108, "reward_meter_mean": 0.668758749961853, "reward_meter_std": 0.44776076078414917, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9480493068695068, "reward_repeat_soft_std": 0.03862285614013672, "reward_judge_quality_mean": 0.26249998807907104, "reward_judge_quality_std": 0.13562026619911194, "reward_total_composite_mean": 0.3588586449623108, "reward_total_composite_std": 0.23403485119342804} {"timestamp_utc": "2026-04-13T11:27:26Z", "mode": "train", "global_step": 1667, "epoch": 0.16745354093420392, "loss": -0.2307, "grad_norm": 1.59761381149292, "learning_rate": 4.951515151515152e-06, "num_tokens": 2957507.0, "completions/mean_length": 206.75, "completions/min_length": 127.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 163.1428680419922, "completions/min_terminated_length": 127.0, "completions/max_terminated_length": 185.0, "rewards/meter/mean": 0.7981754541397095, "rewards/meter/std": 0.328860342502594, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.28171807527542114, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.7329895496368408, "rewards/repeat_soft/std": 0.15630879998207092, "rewards/judge_quality/mean": 0.3187499940395355, "rewards/judge_quality/std": 0.14961259067058563, "rewards/total_composite/mean": 0.4375540316104889, "rewards/total_composite/std": 0.18709851801395416, "reward": 0.4375540316104889, "reward_std": 0.18709851801395416, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06901933997869492, "sampling/sampling_logp_difference/max": 1.4699207544326782, "sampling/importance_sampling_ratio/min": 0.22994370758533478, "sampling/importance_sampling_ratio/mean": 1.0060914754867554, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3053620681166649, "clip_ratio/low_mean": 0.003937007859349251, "clip_ratio/low_min": 0.003937007859349251, "clip_ratio/high_mean": 0.0556541346013546, "clip_ratio/high_max": 0.0556541346013546, "clip_ratio/region_mean": 0.05959114246070385, "reward_total_mean": 0.4375540316104889, "reward_meter_mean": 0.7981754541397095, "reward_meter_std": 0.328860342502594, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.28171807527542114, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.7329895496368408, "reward_repeat_soft_std": 0.15630879998207092, "reward_judge_quality_mean": 0.3187499940395355, "reward_judge_quality_std": 0.14961259067058563, "reward_total_composite_mean": 0.4375540316104889, "reward_total_composite_std": 0.18709851801395416} {"timestamp_utc": "2026-04-13T11:27:38Z", "mode": "train", "global_step": 1668, "epoch": 0.1675539929683576, "loss": -0.2336, "grad_norm": 2.001739740371704, "learning_rate": 4.9484848484848495e-06, "num_tokens": 2959795.0, "completions/mean_length": 285.0, "completions/min_length": 127.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 148.8000030517578, "completions/min_terminated_length": 127.0, "completions/max_terminated_length": 164.0, "rewards/meter/mean": 0.6233569979667664, "rewards/meter/std": 0.3930578827857971, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.2651650309562683, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.9499600529670715, "rewards/repeat_soft/std": 0.0399932935833931, "rewards/judge_quality/mean": 0.20749999582767487, "rewards/judge_quality/std": 0.18312759697437286, "rewards/total_composite/mean": 0.31414613127708435, "rewards/total_composite/std": 0.2726595997810364, "reward": 0.31414613127708435, "reward_std": 0.2726595997810364, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10766993463039398, "sampling/sampling_logp_difference/max": 1.7286872863769531, "sampling/importance_sampling_ratio/min": 0.17751729488372803, "sampling/importance_sampling_ratio/mean": 1.0270744562149048, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6006448864936829, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.06657080911099911, "clip_ratio/high_max": 0.06657080911099911, "clip_ratio/region_mean": 0.06657080911099911, "reward_total_mean": 0.31414613127708435, "reward_meter_mean": 0.6233569979667664, "reward_meter_std": 0.3930578827857971, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.2651650309562683, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.9499600529670715, "reward_repeat_soft_std": 0.0399932935833931, "reward_judge_quality_mean": 0.20749999582767487, "reward_judge_quality_std": 0.18312759697437286, "reward_total_composite_mean": 0.31414613127708435, "reward_total_composite_std": 0.2726595997810364} {"timestamp_utc": "2026-04-13T11:27:50Z", "mode": "train", "global_step": 1669, "epoch": 0.1676544450025113, "loss": -0.08, "grad_norm": 2.2405269145965576, "learning_rate": 4.945454545454546e-06, "num_tokens": 2961143.0, "completions/mean_length": 85.5, "completions/min_length": 18.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 24.571430206298828, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.831732988357544, "rewards/meter/std": 0.3342703580856323, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9585226774215698, "rewards/repeat_soft/std": 0.011249415576457977, "rewards/judge_quality/mean": 0.45125001668930054, "rewards/judge_quality/std": 0.23381540179252625, "rewards/total_composite/mean": 0.5760042071342468, "rewards/total_composite/std": 0.25966617465019226, "reward": 0.5760042071342468, "reward_std": 0.2596661448478699, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11087220162153244, "sampling/sampling_logp_difference/max": 1.0156540870666504, "sampling/importance_sampling_ratio/min": 0.3621654808521271, "sampling/importance_sampling_ratio/mean": 1.0283763408660889, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5842178240418434, "clip_ratio/low_mean": 0.015625, "clip_ratio/low_min": 0.015625, "clip_ratio/high_mean": 0.07428115187212825, "clip_ratio/high_max": 0.07428115187212825, "clip_ratio/region_mean": 0.08990615187212825, "reward_total_mean": 0.5760042071342468, "reward_meter_mean": 0.831732988357544, "reward_meter_std": 0.3342703580856323, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9585226774215698, "reward_repeat_soft_std": 0.011249415576457977, "reward_judge_quality_mean": 0.45125001668930054, "reward_judge_quality_std": 0.23381540179252625, "reward_total_composite_mean": 0.5760042071342468, "reward_total_composite_std": 0.25966617465019226} {"timestamp_utc": "2026-04-13T11:27:56Z", "mode": "train", "global_step": 1670, "epoch": 0.167754897036665, "loss": 0.0438, "grad_norm": 16.120502471923828, "learning_rate": 4.942424242424243e-06, "num_tokens": 2962746.0, "completions/mean_length": 38.375, "completions/min_length": 29.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.375, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.846521258354187, "rewards/meter/std": 0.26176145672798157, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9566450119018555, "rewards/repeat_soft/std": 0.06320735812187195, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.12906257808208466, "rewards/total_composite/mean": 0.5743720531463623, "rewards/total_composite/std": 0.06774458289146423, "reward": 0.5743720531463623, "reward_std": 0.06774458289146423, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1517239809036255, "sampling/sampling_logp_difference/max": 2.793018341064453, "sampling/importance_sampling_ratio/min": 0.061236102133989334, "sampling/importance_sampling_ratio/mean": 0.9983158111572266, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6926508843898773, "clip_ratio/low_mean": 0.025297619868069887, "clip_ratio/low_min": 0.025297619868069887, "clip_ratio/high_mean": 0.09588289353996515, "clip_ratio/high_max": 0.09588289353996515, "clip_ratio/region_mean": 0.12118051340803504, "reward_total_mean": 0.5743720531463623, "reward_meter_mean": 0.846521258354187, "reward_meter_std": 0.26176145672798157, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9566450119018555, "reward_repeat_soft_std": 0.06320735812187195, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.12906257808208466, "reward_total_composite_mean": 0.5743720531463623, "reward_total_composite_std": 0.06774458289146423} {"timestamp_utc": "2026-04-13T11:28:05Z", "mode": "train", "global_step": 1671, "epoch": 0.16785534907081867, "loss": 0.0323, "grad_norm": 5.545763969421387, "learning_rate": 4.93939393939394e-06, "num_tokens": 2965486.0, "completions/mean_length": 159.5, "completions/min_length": 129.0, "completions/max_length": 197.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 159.5, "completions/min_terminated_length": 129.0, "completions/max_terminated_length": 197.0, "rewards/meter/mean": 0.6789463758468628, "rewards/meter/std": 0.3376133143901825, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8708106875419617, "rewards/repeat_soft/std": 0.12946419417858124, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.4800385534763336, "rewards/total_composite/std": 0.07363229244947433, "reward": 0.4800385534763336, "reward_std": 0.07363227754831314, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0955670103430748, "sampling/sampling_logp_difference/max": 1.5123744010925293, "sampling/importance_sampling_ratio/min": 0.2203860729932785, "sampling/importance_sampling_ratio/mean": 1.004928469657898, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.565850742161274, "clip_ratio/low_mean": 0.04794883681461215, "clip_ratio/low_min": 0.04794883681461215, "clip_ratio/high_mean": 0.035761090926826, "clip_ratio/high_max": 0.035761090926826, "clip_ratio/region_mean": 0.08370992774143815, "reward_total_mean": 0.4800385534763336, "reward_meter_mean": 0.6789463758468628, "reward_meter_std": 0.3376133143901825, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8708106875419617, "reward_repeat_soft_std": 0.12946419417858124, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.4800385534763336, "reward_total_composite_std": 0.07363229244947433} {"timestamp_utc": "2026-04-13T11:28:11Z", "mode": "train", "global_step": 1672, "epoch": 0.16795580110497238, "loss": 0.2415, "grad_norm": 18.02611541748047, "learning_rate": 4.936363636363637e-06, "num_tokens": 2966841.0, "completions/mean_length": 26.375, "completions/min_length": 20.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.375, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.8848724961280823, "rewards/meter/std": 0.294619083404541, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.713463306427002, "rewards/repeat_soft/std": 0.34030842781066895, "rewards/judge_quality/mean": 0.4724999964237213, "rewards/judge_quality/std": 0.19255799055099487, "rewards/total_composite/mean": 0.5555944442749023, "rewards/total_composite/std": 0.20079876482486725, "reward": 0.5555944442749023, "reward_std": 0.20079874992370605, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11104923486709595, "sampling/sampling_logp_difference/max": 1.8065016269683838, "sampling/importance_sampling_ratio/min": 0.1642276644706726, "sampling/importance_sampling_ratio/mean": 1.0184675455093384, "sampling/importance_sampling_ratio/max": 1.7557411193847656, "entropy": 0.7442312054336071, "clip_ratio/low_mean": 0.029363354202359915, "clip_ratio/low_min": 0.029363354202359915, "clip_ratio/high_mean": 0.05451839976012707, "clip_ratio/high_max": 0.05451839976012707, "clip_ratio/region_mean": 0.08388175396248698, "reward_total_mean": 0.5555944442749023, "reward_meter_mean": 0.8848724961280823, "reward_meter_std": 0.294619083404541, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.713463306427002, "reward_repeat_soft_std": 0.34030842781066895, "reward_judge_quality_mean": 0.4724999964237213, "reward_judge_quality_std": 0.19255799055099487, "reward_total_composite_mean": 0.5555944442749023, "reward_total_composite_std": 0.20079876482486725} {"timestamp_utc": "2026-04-13T11:28:18Z", "mode": "train", "global_step": 1673, "epoch": 0.16805625313912606, "loss": -0.0, "grad_norm": 7.407658100128174, "learning_rate": 4.933333333333334e-06, "num_tokens": 2968831.0, "completions/mean_length": 80.75, "completions/min_length": 64.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.75, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.6783466339111328, "rewards/meter/std": 0.36895039677619934, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.720346212387085, "rewards/repeat_soft/std": 0.13622376322746277, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.49563726782798767, "rewards/total_composite/std": 0.09533317387104034, "reward": 0.49563726782798767, "reward_std": 0.09533317387104034, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0666607990860939, "sampling/sampling_logp_difference/max": 1.689176321029663, "sampling/importance_sampling_ratio/min": 0.1846715807914734, "sampling/importance_sampling_ratio/mean": 1.0037403106689453, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3736836090683937, "clip_ratio/low_mean": 0.023216268629767, "clip_ratio/low_min": 0.023216268629767, "clip_ratio/high_mean": 0.036169617902487516, "clip_ratio/high_max": 0.036169617902487516, "clip_ratio/region_mean": 0.05938588653225452, "reward_total_mean": 0.49563726782798767, "reward_meter_mean": 0.6783466339111328, "reward_meter_std": 0.36895039677619934, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.720346212387085, "reward_repeat_soft_std": 0.13622376322746277, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.49563726782798767, "reward_total_composite_std": 0.09533317387104034} {"timestamp_utc": "2026-04-13T11:28:25Z", "mode": "train", "global_step": 1674, "epoch": 0.16815670517327977, "loss": 0.021, "grad_norm": 8.999605178833008, "learning_rate": 4.9303030303030305e-06, "num_tokens": 2970778.0, "completions/mean_length": 84.375, "completions/min_length": 62.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.375, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.32025760412216187, "rewards/meter/std": 0.1794765591621399, "rewards/count_adherence/mean": 0.7083333730697632, "rewards/count_adherence/std": 0.11785111576318741, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8883340358734131, "rewards/repeat_soft/std": 0.07298793643712997, "rewards/judge_quality/mean": 0.5824999809265137, "rewards/judge_quality/std": 0.23260943591594696, "rewards/total_composite/mean": 0.4041918218135834, "rewards/total_composite/std": 0.1147385984659195, "reward": 0.4041918218135834, "reward_std": 0.11473860591650009, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11148685216903687, "sampling/sampling_logp_difference/max": 2.3669281005859375, "sampling/importance_sampling_ratio/min": 0.09376832842826843, "sampling/importance_sampling_ratio/mean": 0.9885243773460388, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.39864677004516125, "clip_ratio/low_mean": 0.0698322867974639, "clip_ratio/low_min": 0.0698322867974639, "clip_ratio/high_mean": 0.03228022065013647, "clip_ratio/high_max": 0.03228022065013647, "clip_ratio/region_mean": 0.10211250744760036, "reward_total_mean": 0.4041918218135834, "reward_meter_mean": 0.32025760412216187, "reward_meter_std": 0.1794765591621399, "reward_count_adherence_mean": 0.7083333730697632, "reward_count_adherence_std": 0.11785111576318741, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8883340358734131, "reward_repeat_soft_std": 0.07298793643712997, "reward_judge_quality_mean": 0.5824999809265137, "reward_judge_quality_std": 0.23260943591594696, "reward_total_composite_mean": 0.4041918218135834, "reward_total_composite_std": 0.1147385984659195} {"timestamp_utc": "2026-04-13T11:28:33Z", "mode": "train", "global_step": 1675, "epoch": 0.16825715720743345, "loss": -0.0038, "grad_norm": 4.344687461853027, "learning_rate": 4.927272727272728e-06, "num_tokens": 2973342.0, "completions/mean_length": 126.5, "completions/min_length": 114.0, "completions/max_length": 141.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 126.5, "completions/min_terminated_length": 114.0, "completions/max_terminated_length": 141.0, "rewards/meter/mean": 0.8118463754653931, "rewards/meter/std": 0.30178484320640564, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6909312605857849, "rewards/repeat_soft/std": 0.048823073506355286, "rewards/judge_quality/mean": 0.4024999737739563, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.5161367654800415, "rewards/total_composite/std": 0.09046050906181335, "reward": 0.5161367654800415, "reward_std": 0.09046050906181335, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05651190131902695, "sampling/sampling_logp_difference/max": 1.9631762504577637, "sampling/importance_sampling_ratio/min": 0.14041173458099365, "sampling/importance_sampling_ratio/mean": 1.006211280822754, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.301896084100008, "clip_ratio/low_mean": 0.013461179565638304, "clip_ratio/low_min": 0.013461179565638304, "clip_ratio/high_mean": 0.0389620391651988, "clip_ratio/high_max": 0.0389620391651988, "clip_ratio/region_mean": 0.05242321873083711, "reward_total_mean": 0.5161367654800415, "reward_meter_mean": 0.8118463754653931, "reward_meter_std": 0.30178484320640564, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6909312605857849, "reward_repeat_soft_std": 0.048823073506355286, "reward_judge_quality_mean": 0.4024999737739563, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.5161367654800415, "reward_total_composite_std": 0.09046050906181335} {"timestamp_utc": "2026-04-13T11:28:42Z", "mode": "train", "global_step": 1676, "epoch": 0.16835760924158713, "loss": 0.0526, "grad_norm": 4.972901821136475, "learning_rate": 4.924242424242425e-06, "num_tokens": 2976235.0, "completions/mean_length": 161.625, "completions/min_length": 131.0, "completions/max_length": 212.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 161.625, "completions/min_terminated_length": 131.0, "completions/max_terminated_length": 212.0, "rewards/meter/mean": 0.8618706464767456, "rewards/meter/std": 0.2439996600151062, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7606981992721558, "rewards/repeat_soft/std": 0.13278447091579437, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.503989040851593, "rewards/total_composite/std": 0.06762686371803284, "reward": 0.503989040851593, "reward_std": 0.06762687116861343, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07873500883579254, "sampling/sampling_logp_difference/max": 1.9043989181518555, "sampling/importance_sampling_ratio/min": 0.1527395099401474, "sampling/importance_sampling_ratio/mean": 1.0007163286209106, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.40678513795137405, "clip_ratio/low_mean": 0.022617980604991317, "clip_ratio/low_min": 0.022617980604991317, "clip_ratio/high_mean": 0.045249105896800756, "clip_ratio/high_max": 0.045249105896800756, "clip_ratio/region_mean": 0.06786708650179207, "reward_total_mean": 0.503989040851593, "reward_meter_mean": 0.8618706464767456, "reward_meter_std": 0.2439996600151062, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7606981992721558, "reward_repeat_soft_std": 0.13278447091579437, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.503989040851593, "reward_total_composite_std": 0.06762686371803284} {"timestamp_utc": "2026-04-13T11:28:48Z", "mode": "train", "global_step": 1677, "epoch": 0.16845806127574084, "loss": -0.0091, "grad_norm": 9.968160629272461, "learning_rate": 4.9212121212121214e-06, "num_tokens": 2977843.0, "completions/mean_length": 40.0, "completions/min_length": 35.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.5374641418457031, "rewards/meter/std": 0.4309203624725342, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8733203411102295, "rewards/repeat_soft/std": 0.058413829654455185, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.4737306237220764, "rewards/total_composite/std": 0.26079532504081726, "reward": 0.4737306237220764, "reward_std": 0.2607952952384949, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08989720791578293, "sampling/sampling_logp_difference/max": 2.8600518703460693, "sampling/importance_sampling_ratio/min": 0.057265788316726685, "sampling/importance_sampling_ratio/mean": 1.0158300399780273, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.49360642209649086, "clip_ratio/low_mean": 0.026151658268645406, "clip_ratio/low_min": 0.026151658268645406, "clip_ratio/high_mean": 0.03813495417125523, "clip_ratio/high_max": 0.03813495417125523, "clip_ratio/region_mean": 0.06428661243990064, "reward_total_mean": 0.4737306237220764, "reward_meter_mean": 0.5374641418457031, "reward_meter_std": 0.4309203624725342, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8733203411102295, "reward_repeat_soft_std": 0.058413829654455185, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.4737306237220764, "reward_total_composite_std": 0.26079532504081726} {"timestamp_utc": "2026-04-13T11:28:57Z", "mode": "train", "global_step": 1678, "epoch": 0.16855851330989452, "loss": -0.0214, "grad_norm": 2.942268133163452, "learning_rate": 4.918181818181819e-06, "num_tokens": 2980783.0, "completions/mean_length": 186.5, "completions/min_length": 154.0, "completions/max_length": 226.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 186.5, "completions/min_terminated_length": 154.0, "completions/max_terminated_length": 226.0, "rewards/meter/mean": 0.8959339261054993, "rewards/meter/std": 0.21870744228363037, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.5695617198944092, "rewards/repeat_soft/std": 0.0780857726931572, "rewards/judge_quality/mean": 0.22499999403953552, "rewards/judge_quality/std": 0.04629100486636162, "rewards/total_composite/mean": 0.4149327278137207, "rewards/total_composite/std": 0.04342221841216087, "reward": 0.4149327278137207, "reward_std": 0.04342221841216087, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04488222673535347, "sampling/sampling_logp_difference/max": 2.6392836570739746, "sampling/importance_sampling_ratio/min": 0.07141240686178207, "sampling/importance_sampling_ratio/mean": 1.0022094249725342, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.274970768019557, "clip_ratio/low_mean": 0.011812139535322785, "clip_ratio/low_min": 0.011812139535322785, "clip_ratio/high_mean": 0.03270714543759823, "clip_ratio/high_max": 0.03270714543759823, "clip_ratio/region_mean": 0.044519284972921014, "reward_total_mean": 0.4149327278137207, "reward_meter_mean": 0.8959339261054993, "reward_meter_std": 0.21870744228363037, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.5695617198944092, "reward_repeat_soft_std": 0.0780857726931572, "reward_judge_quality_mean": 0.22499999403953552, "reward_judge_quality_std": 0.04629100486636162, "reward_total_composite_mean": 0.4149327278137207, "reward_total_composite_std": 0.04342221841216087} {"timestamp_utc": "2026-04-13T11:29:08Z", "mode": "train", "global_step": 1679, "epoch": 0.16865896534404823, "loss": -0.2272, "grad_norm": 1.6492377519607544, "learning_rate": 4.915151515151516e-06, "num_tokens": 2983300.0, "completions/mean_length": 199.625, "completions/min_length": 99.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 155.0, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 187.0, "rewards/meter/mean": 0.769597589969635, "rewards/meter/std": 0.20830576121807098, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.21380899846553802, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.6938011646270752, "rewards/repeat_soft/std": 0.13908973336219788, "rewards/judge_quality/mean": 0.2887499928474426, "rewards/judge_quality/std": 0.15797263383865356, "rewards/total_composite/mean": 0.39985230565071106, "rewards/total_composite/std": 0.17132367193698883, "reward": 0.39985230565071106, "reward_std": 0.17132367193698883, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0714179128408432, "sampling/sampling_logp_difference/max": 3.210151195526123, "sampling/importance_sampling_ratio/min": 0.04035051539540291, "sampling/importance_sampling_ratio/mean": 1.013409972190857, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.30607396736741066, "clip_ratio/low_mean": 0.0033783784601837397, "clip_ratio/low_min": 0.0033783784601837397, "clip_ratio/high_mean": 0.045120069524273276, "clip_ratio/high_max": 0.045120069524273276, "clip_ratio/region_mean": 0.048498447984457016, "reward_total_mean": 0.39985230565071106, "reward_meter_mean": 0.769597589969635, "reward_meter_std": 0.20830576121807098, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.21380899846553802, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.6938011646270752, "reward_repeat_soft_std": 0.13908973336219788, "reward_judge_quality_mean": 0.2887499928474426, "reward_judge_quality_std": 0.15797263383865356, "reward_total_composite_mean": 0.39985230565071106, "reward_total_composite_std": 0.17132367193698883} {"timestamp_utc": "2026-04-13T11:29:15Z", "mode": "train", "global_step": 1680, "epoch": 0.1687594173782019, "loss": 0.077, "grad_norm": 9.330824851989746, "learning_rate": 4.912121212121212e-06, "num_tokens": 2985351.0, "completions/mean_length": 94.375, "completions/min_length": 82.0, "completions/max_length": 111.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.375, "completions/min_terminated_length": 82.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.8564003109931946, "rewards/meter/std": 0.1464369297027588, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7046080827713013, "rewards/repeat_soft/std": 0.07144229859113693, "rewards/judge_quality/mean": 0.38624998927116394, "rewards/judge_quality/std": 0.09545940905809402, "rewards/total_composite/mean": 0.470175176858902, "rewards/total_composite/std": 0.06205389276146889, "reward": 0.470175176858902, "reward_std": 0.062053896486759186, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08212559670209885, "sampling/sampling_logp_difference/max": 2.9680633544921875, "sampling/importance_sampling_ratio/min": 0.05140276253223419, "sampling/importance_sampling_ratio/mean": 0.9970459342002869, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.302042568102479, "clip_ratio/low_mean": 0.011065413244068623, "clip_ratio/low_min": 0.011065413244068623, "clip_ratio/high_mean": 0.05521555768791586, "clip_ratio/high_max": 0.05521555768791586, "clip_ratio/region_mean": 0.06628097093198448, "reward_total_mean": 0.470175176858902, "reward_meter_mean": 0.8564003109931946, "reward_meter_std": 0.1464369297027588, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7046080827713013, "reward_repeat_soft_std": 0.07144229859113693, "reward_judge_quality_mean": 0.38624998927116394, "reward_judge_quality_std": 0.09545940905809402, "reward_total_composite_mean": 0.470175176858902, "reward_total_composite_std": 0.06205389276146889} {"timestamp_utc": "2026-04-13T11:29:22Z", "mode": "train", "global_step": 1681, "epoch": 0.1688598694123556, "loss": 0.0655, "grad_norm": 9.672286987304688, "learning_rate": 4.90909090909091e-06, "num_tokens": 2987043.0, "completions/mean_length": 56.5, "completions/min_length": 45.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.5, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9728363156318665, "rewards/meter/std": 0.018391400575637817, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8520519137382507, "rewards/repeat_soft/std": 0.04583599418401718, "rewards/judge_quality/mean": 0.32624998688697815, "rewards/judge_quality/std": 0.11350739002227783, "rewards/total_composite/mean": 0.5336079597473145, "rewards/total_composite/std": 0.07198585569858551, "reward": 0.5336079597473145, "reward_std": 0.07198584824800491, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0933646410703659, "sampling/sampling_logp_difference/max": 1.170951008796692, "sampling/importance_sampling_ratio/min": 0.3100719451904297, "sampling/importance_sampling_ratio/mean": 1.0184705257415771, "sampling/importance_sampling_ratio/max": 1.6749534606933594, "entropy": 0.626958291977644, "clip_ratio/low_mean": 0.032328391214832664, "clip_ratio/low_min": 0.032328391214832664, "clip_ratio/high_mean": 0.0491899773478508, "clip_ratio/high_max": 0.0491899773478508, "clip_ratio/region_mean": 0.08151836856268346, "reward_total_mean": 0.5336079597473145, "reward_meter_mean": 0.9728363156318665, "reward_meter_std": 0.018391400575637817, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8520519137382507, "reward_repeat_soft_std": 0.04583599418401718, "reward_judge_quality_mean": 0.32624998688697815, "reward_judge_quality_std": 0.11350739002227783, "reward_total_composite_mean": 0.5336079597473145, "reward_total_composite_std": 0.07198585569858551} {"timestamp_utc": "2026-04-13T11:29:28Z", "mode": "train", "global_step": 1682, "epoch": 0.1689603214465093, "loss": 0.041, "grad_norm": 12.904158592224121, "learning_rate": 4.906060606060606e-06, "num_tokens": 2988684.0, "completions/mean_length": 54.125, "completions/min_length": 49.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.125, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9888840317726135, "rewards/meter/std": 0.013273422606289387, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8677555322647095, "rewards/repeat_soft/std": 0.07466918230056763, "rewards/judge_quality/mean": 0.4025000035762787, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.5888320207595825, "rewards/total_composite/std": 0.04263556748628616, "reward": 0.5888320207595825, "reward_std": 0.04263556748628616, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09531761705875397, "sampling/sampling_logp_difference/max": 1.3383307456970215, "sampling/importance_sampling_ratio/min": 0.2622831165790558, "sampling/importance_sampling_ratio/mean": 1.0110734701156616, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5609508864581585, "clip_ratio/low_mean": 0.028822056017816067, "clip_ratio/low_min": 0.028822056017816067, "clip_ratio/high_mean": 0.06253244495019317, "clip_ratio/high_max": 0.06253244495019317, "clip_ratio/region_mean": 0.09135450096800923, "reward_total_mean": 0.5888320207595825, "reward_meter_mean": 0.9888840317726135, "reward_meter_std": 0.013273422606289387, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8677555322647095, "reward_repeat_soft_std": 0.07466918230056763, "reward_judge_quality_mean": 0.4025000035762787, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.5888320207595825, "reward_total_composite_std": 0.04263556748628616} {"timestamp_utc": "2026-04-13T11:29:35Z", "mode": "train", "global_step": 1683, "epoch": 0.16906077348066298, "loss": 0.0611, "grad_norm": 12.600188255310059, "learning_rate": 4.903030303030303e-06, "num_tokens": 2990363.0, "completions/mean_length": 51.875, "completions/min_length": 42.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.875, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.49862271547317505, "rewards/meter/std": 0.32598933577537537, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9934655427932739, "rewards/repeat_soft/std": 0.011311059817671776, "rewards/judge_quality/mean": 0.6349999904632568, "rewards/judge_quality/std": 0.2235429286956787, "rewards/total_composite/mean": 0.5552878379821777, "rewards/total_composite/std": 0.13822627067565918, "reward": 0.5552878379821777, "reward_std": 0.13822627067565918, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12635992467403412, "sampling/sampling_logp_difference/max": 1.9483773708343506, "sampling/importance_sampling_ratio/min": 0.14250512421131134, "sampling/importance_sampling_ratio/mean": 1.0187221765518188, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.514126792550087, "clip_ratio/low_mean": 0.05596538633108139, "clip_ratio/low_min": 0.05596538633108139, "clip_ratio/high_mean": 0.04068396287038922, "clip_ratio/high_max": 0.04068396287038922, "clip_ratio/region_mean": 0.09664934920147061, "reward_total_mean": 0.5552878379821777, "reward_meter_mean": 0.49862271547317505, "reward_meter_std": 0.32598933577537537, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9934655427932739, "reward_repeat_soft_std": 0.011311059817671776, "reward_judge_quality_mean": 0.6349999904632568, "reward_judge_quality_std": 0.2235429286956787, "reward_total_composite_mean": 0.5552878379821777, "reward_total_composite_std": 0.13822627067565918} {"timestamp_utc": "2026-04-13T11:29:41Z", "mode": "train", "global_step": 1684, "epoch": 0.1691612255148167, "loss": 0.1425, "grad_norm": 13.452690124511719, "learning_rate": 4.9000000000000005e-06, "num_tokens": 2991738.0, "completions/mean_length": 32.875, "completions/min_length": 20.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.875, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.6996747255325317, "rewards/meter/std": 0.4190917909145355, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.4629100561141968, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7916808128356934, "rewards/repeat_soft/std": 0.1934211254119873, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.12972497940063477, "rewards/total_composite/mean": 0.4318060874938965, "rewards/total_composite/std": 0.12287203222513199, "reward": 0.4318060874938965, "reward_std": 0.12287202477455139, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12454545497894287, "sampling/sampling_logp_difference/max": 2.507312774658203, "sampling/importance_sampling_ratio/min": 0.08148691803216934, "sampling/importance_sampling_ratio/mean": 0.9963033199310303, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7385776005685329, "clip_ratio/low_mean": 0.04498840123414993, "clip_ratio/low_min": 0.04498840123414993, "clip_ratio/high_mean": 0.053281772416085005, "clip_ratio/high_max": 0.053281772416085005, "clip_ratio/region_mean": 0.09827017365023494, "reward_total_mean": 0.4318060874938965, "reward_meter_mean": 0.6996747255325317, "reward_meter_std": 0.4190917909145355, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.4629100561141968, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7916808128356934, "reward_repeat_soft_std": 0.1934211254119873, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.12972497940063477, "reward_total_composite_mean": 0.4318060874938965, "reward_total_composite_std": 0.12287203222513199} {"timestamp_utc": "2026-04-13T11:29:49Z", "mode": "train", "global_step": 1685, "epoch": 0.16926167754897037, "loss": 0.0517, "grad_norm": 6.260843276977539, "learning_rate": 4.896969696969697e-06, "num_tokens": 2994191.0, "completions/mean_length": 130.625, "completions/min_length": 109.0, "completions/max_length": 151.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 130.625, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 151.0, "rewards/meter/mean": 0.8883123397827148, "rewards/meter/std": 0.19294770061969757, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9430127143859863, "rewards/repeat_soft/std": 0.026210656389594078, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.10966669768095016, "rewards/total_composite/mean": 0.5498702526092529, "rewards/total_composite/std": 0.08809173852205276, "reward": 0.5498702526092529, "reward_std": 0.08809174597263336, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1032494604587555, "sampling/sampling_logp_difference/max": 1.8032281398773193, "sampling/importance_sampling_ratio/min": 0.1647661328315735, "sampling/importance_sampling_ratio/mean": 1.0128101110458374, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6937706656754017, "clip_ratio/low_mean": 0.02845947165042162, "clip_ratio/low_min": 0.02845947165042162, "clip_ratio/high_mean": 0.06688423454761505, "clip_ratio/high_max": 0.06688423454761505, "clip_ratio/region_mean": 0.09534370619803667, "reward_total_mean": 0.5498702526092529, "reward_meter_mean": 0.8883123397827148, "reward_meter_std": 0.19294770061969757, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9430127143859863, "reward_repeat_soft_std": 0.026210656389594078, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.10966669768095016, "reward_total_composite_mean": 0.5498702526092529, "reward_total_composite_std": 0.08809173852205276} {"timestamp_utc": "2026-04-13T11:29:57Z", "mode": "train", "global_step": 1686, "epoch": 0.16936212958312405, "loss": -0.0353, "grad_norm": 5.953235626220703, "learning_rate": 4.893939393939394e-06, "num_tokens": 2996821.0, "completions/mean_length": 132.75, "completions/min_length": 111.0, "completions/max_length": 147.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 132.75, "completions/min_terminated_length": 111.0, "completions/max_terminated_length": 147.0, "rewards/meter/mean": 0.8653354644775391, "rewards/meter/std": 0.222174271941185, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8975305557250977, "rewards/repeat_soft/std": 0.09665177762508392, "rewards/judge_quality/mean": 0.30124998092651367, "rewards/judge_quality/std": 0.10398317128419876, "rewards/total_composite/mean": 0.507197916507721, "rewards/total_composite/std": 0.08576513826847076, "reward": 0.507197916507721, "reward_std": 0.08576513826847076, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0919460579752922, "sampling/sampling_logp_difference/max": 3.7899551391601562, "sampling/importance_sampling_ratio/min": 0.02259661629796028, "sampling/importance_sampling_ratio/mean": 1.0128374099731445, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5677018612623215, "clip_ratio/low_mean": 0.05248046666383743, "clip_ratio/low_min": 0.05248046666383743, "clip_ratio/high_mean": 0.03497450239956379, "clip_ratio/high_max": 0.03497450239956379, "clip_ratio/region_mean": 0.08745496906340122, "reward_total_mean": 0.507197916507721, "reward_meter_mean": 0.8653354644775391, "reward_meter_std": 0.222174271941185, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8975305557250977, "reward_repeat_soft_std": 0.09665177762508392, "reward_judge_quality_mean": 0.30124998092651367, "reward_judge_quality_std": 0.10398317128419876, "reward_total_composite_mean": 0.507197916507721, "reward_total_composite_std": 0.08576513826847076} {"timestamp_utc": "2026-04-13T11:30:03Z", "mode": "train", "global_step": 1687, "epoch": 0.16946258161727776, "loss": 0.0425, "grad_norm": 9.852559089660645, "learning_rate": 4.8909090909090914e-06, "num_tokens": 2998704.0, "completions/mean_length": 64.375, "completions/min_length": 51.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.375, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9665820598602295, "rewards/meter/std": 0.06644105166196823, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9778945446014404, "rewards/repeat_soft/std": 0.03151845559477806, "rewards/judge_quality/mean": 0.4137499928474426, "rewards/judge_quality/std": 0.06781013309955597, "rewards/total_composite/mean": 0.6065137386322021, "rewards/total_composite/std": 0.046656109392642975, "reward": 0.6065137386322021, "reward_std": 0.04665609821677208, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11938251554965973, "sampling/sampling_logp_difference/max": 2.54801869392395, "sampling/importance_sampling_ratio/min": 0.07823652774095535, "sampling/importance_sampling_ratio/mean": 1.0205943584442139, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8639738485217094, "clip_ratio/low_mean": 0.05129922367632389, "clip_ratio/low_min": 0.05129922367632389, "clip_ratio/high_mean": 0.06819419329985976, "clip_ratio/high_max": 0.06819419329985976, "clip_ratio/region_mean": 0.11949341697618365, "reward_total_mean": 0.6065137386322021, "reward_meter_mean": 0.9665820598602295, "reward_meter_std": 0.06644105166196823, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9778945446014404, "reward_repeat_soft_std": 0.03151845559477806, "reward_judge_quality_mean": 0.4137499928474426, "reward_judge_quality_std": 0.06781013309955597, "reward_total_composite_mean": 0.6065137386322021, "reward_total_composite_std": 0.046656109392642975} {"timestamp_utc": "2026-04-13T11:30:11Z", "mode": "train", "global_step": 1688, "epoch": 0.16956303365143144, "loss": -0.0767, "grad_norm": 3.8066842555999756, "learning_rate": 4.887878787878788e-06, "num_tokens": 3001705.0, "completions/mean_length": 177.125, "completions/min_length": 130.0, "completions/max_length": 207.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 177.125, "completions/min_terminated_length": 130.0, "completions/max_terminated_length": 207.0, "rewards/meter/mean": 0.9835997223854065, "rewards/meter/std": 0.01157785952091217, "rewards/count_adherence/mean": 0.9583333134651184, "rewards/count_adherence/std": 0.07715168595314026, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6306673884391785, "rewards/repeat_soft/std": 0.13541024923324585, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.5140981674194336, "rewards/total_composite/std": 0.07761696726083755, "reward": 0.5140981674194336, "reward_std": 0.07761697471141815, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0640011578798294, "sampling/sampling_logp_difference/max": 2.27779483795166, "sampling/importance_sampling_ratio/min": 0.1025100126862526, "sampling/importance_sampling_ratio/mean": 1.0001037120819092, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3475971445441246, "clip_ratio/low_mean": 0.016462384024634957, "clip_ratio/low_min": 0.016462384024634957, "clip_ratio/high_mean": 0.040552848717197776, "clip_ratio/high_max": 0.040552848717197776, "clip_ratio/region_mean": 0.05701523274183273, "reward_total_mean": 0.5140981674194336, "reward_meter_mean": 0.9835997223854065, "reward_meter_std": 0.01157785952091217, "reward_count_adherence_mean": 0.9583333134651184, "reward_count_adherence_std": 0.07715168595314026, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6306673884391785, "reward_repeat_soft_std": 0.13541024923324585, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.5140981674194336, "reward_total_composite_std": 0.07761696726083755} {"timestamp_utc": "2026-04-13T11:30:18Z", "mode": "train", "global_step": 1689, "epoch": 0.16966348568558512, "loss": 0.0366, "grad_norm": 10.181629180908203, "learning_rate": 4.884848484848485e-06, "num_tokens": 3003449.0, "completions/mean_length": 59.0, "completions/min_length": 53.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.0, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.8670849204063416, "rewards/meter/std": 0.30219969153404236, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9106879234313965, "rewards/repeat_soft/std": 0.030124902725219727, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.5828097462654114, "rewards/total_composite/std": 0.09014017134904861, "reward": 0.5828097462654114, "reward_std": 0.09014015644788742, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11158782988786697, "sampling/sampling_logp_difference/max": 1.7126024961471558, "sampling/importance_sampling_ratio/min": 0.18039570748806, "sampling/importance_sampling_ratio/mean": 1.0069172382354736, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5254568979144096, "clip_ratio/low_mean": 0.01411290280520916, "clip_ratio/low_min": 0.01411290280520916, "clip_ratio/high_mean": 0.08197389869019389, "clip_ratio/high_max": 0.08197389869019389, "clip_ratio/region_mean": 0.09608680149540305, "reward_total_mean": 0.5828097462654114, "reward_meter_mean": 0.8670849204063416, "reward_meter_std": 0.30219969153404236, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9106879234313965, "reward_repeat_soft_std": 0.030124902725219727, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.5828097462654114, "reward_total_composite_std": 0.09014017134904861} {"timestamp_utc": "2026-04-13T11:30:24Z", "mode": "train", "global_step": 1690, "epoch": 0.16976393771973883, "loss": 0.0718, "grad_norm": 17.32733917236328, "learning_rate": 4.881818181818182e-06, "num_tokens": 3004870.0, "completions/mean_length": 19.625, "completions/min_length": 17.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 19.625, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.9793528318405151, "rewards/meter/std": 0.013812100514769554, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8885176777839661, "rewards/repeat_soft/std": 0.06772414594888687, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.6445207595825195, "rewards/total_composite/std": 0.09985730051994324, "reward": 0.6445207595825195, "reward_std": 0.09985730797052383, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12900254130363464, "sampling/sampling_logp_difference/max": 1.0521681308746338, "sampling/importance_sampling_ratio/min": 0.34917983412742615, "sampling/importance_sampling_ratio/mean": 1.0159752368927002, "sampling/importance_sampling_ratio/max": 1.860042691230774, "entropy": 0.7945980131626129, "clip_ratio/low_mean": 0.12428737711161375, "clip_ratio/low_min": 0.12428737711161375, "clip_ratio/high_mean": 0.022058824077248573, "clip_ratio/high_max": 0.022058824077248573, "clip_ratio/region_mean": 0.14634620118886232, "reward_total_mean": 0.6445207595825195, "reward_meter_mean": 0.9793528318405151, "reward_meter_std": 0.013812100514769554, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8885176777839661, "reward_repeat_soft_std": 0.06772414594888687, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.6445207595825195, "reward_total_composite_std": 0.09985730051994324} {"timestamp_utc": "2026-04-13T11:30:32Z", "mode": "train", "global_step": 1691, "epoch": 0.1698643897538925, "loss": 0.0555, "grad_norm": 12.87840747833252, "learning_rate": 4.878787878787879e-06, "num_tokens": 3006625.0, "completions/mean_length": 61.375, "completions/min_length": 39.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.375, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.8104844093322754, "rewards/meter/std": 0.19354458153247833, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9004038572311401, "rewards/repeat_soft/std": 0.0796278715133667, "rewards/judge_quality/mean": 0.4987500011920929, "rewards/judge_quality/std": 0.21357084810733795, "rewards/total_composite/mean": 0.5965396165847778, "rewards/total_composite/std": 0.11422868072986603, "reward": 0.5965396165847778, "reward_std": 0.11422866582870483, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13484059274196625, "sampling/sampling_logp_difference/max": 2.514012098312378, "sampling/importance_sampling_ratio/min": 0.08094283193349838, "sampling/importance_sampling_ratio/mean": 1.0066410303115845, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6124727204442024, "clip_ratio/low_mean": 0.09606262110173702, "clip_ratio/low_min": 0.09606262110173702, "clip_ratio/high_mean": 0.042786071076989174, "clip_ratio/high_max": 0.042786071076989174, "clip_ratio/region_mean": 0.1388486921787262, "reward_total_mean": 0.5965396165847778, "reward_meter_mean": 0.8104844093322754, "reward_meter_std": 0.19354458153247833, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9004038572311401, "reward_repeat_soft_std": 0.0796278715133667, "reward_judge_quality_mean": 0.4987500011920929, "reward_judge_quality_std": 0.21357084810733795, "reward_total_composite_mean": 0.5965396165847778, "reward_total_composite_std": 0.11422868072986603} {"timestamp_utc": "2026-04-13T11:30:38Z", "mode": "train", "global_step": 1692, "epoch": 0.16996484178804622, "loss": 0.0305, "grad_norm": 6.780965805053711, "learning_rate": 4.875757575757576e-06, "num_tokens": 3008567.0, "completions/mean_length": 73.75, "completions/min_length": 69.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.75, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.9741885662078857, "rewards/meter/std": 0.01744021102786064, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6959117650985718, "rewards/repeat_soft/std": 0.1492699682712555, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.10966669768095016, "rewards/total_composite/mean": 0.5219532251358032, "rewards/total_composite/std": 0.08748581260442734, "reward": 0.5219532251358032, "reward_std": 0.08748582005500793, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06695970892906189, "sampling/sampling_logp_difference/max": 1.565828800201416, "sampling/importance_sampling_ratio/min": 0.20891478657722473, "sampling/importance_sampling_ratio/mean": 0.9970290064811707, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3333785180002451, "clip_ratio/low_mean": 0.004982864367775619, "clip_ratio/low_min": 0.004982864367775619, "clip_ratio/high_mean": 0.03747419686987996, "clip_ratio/high_max": 0.03747419686987996, "clip_ratio/region_mean": 0.04245706123765558, "reward_total_mean": 0.5219532251358032, "reward_meter_mean": 0.9741885662078857, "reward_meter_std": 0.01744021102786064, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6959117650985718, "reward_repeat_soft_std": 0.1492699682712555, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.10966669768095016, "reward_total_composite_mean": 0.5219532251358032, "reward_total_composite_std": 0.08748581260442734} {"timestamp_utc": "2026-04-13T11:30:49Z", "mode": "train", "global_step": 1693, "epoch": 0.1700652938221999, "loss": -0.1995, "grad_norm": 2.3482632637023926, "learning_rate": 4.872727272727273e-06, "num_tokens": 3011066.0, "completions/mean_length": 174.375, "completions/min_length": 110.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 126.14286041259766, "completions/min_terminated_length": 110.0, "completions/max_terminated_length": 137.0, "rewards/meter/mean": 0.9050018787384033, "rewards/meter/std": 0.20063656568527222, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.7749311923980713, "rewards/repeat_soft/std": 0.09003739804029465, "rewards/judge_quality/mean": 0.3187499940395355, "rewards/judge_quality/std": 0.14961259067058563, "rewards/total_composite/mean": 0.43696027994155884, "rewards/total_composite/std": 0.19049564003944397, "reward": 0.43696027994155884, "reward_std": 0.19049562513828278, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08963243663311005, "sampling/sampling_logp_difference/max": 1.7019519805908203, "sampling/importance_sampling_ratio/min": 0.1823272705078125, "sampling/importance_sampling_ratio/mean": 1.0067193508148193, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.430629450827837, "clip_ratio/low_mean": 0.010019906796514988, "clip_ratio/low_min": 0.010019906796514988, "clip_ratio/high_mean": 0.04426609817892313, "clip_ratio/high_max": 0.04426609817892313, "clip_ratio/region_mean": 0.05428600497543812, "reward_total_mean": 0.43696027994155884, "reward_meter_mean": 0.9050018787384033, "reward_meter_std": 0.20063656568527222, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.7749311923980713, "reward_repeat_soft_std": 0.09003739804029465, "reward_judge_quality_mean": 0.3187499940395355, "reward_judge_quality_std": 0.14961259067058563, "reward_total_composite_mean": 0.43696027994155884, "reward_total_composite_std": 0.19049564003944397} {"timestamp_utc": "2026-04-13T11:30:56Z", "mode": "train", "global_step": 1694, "epoch": 0.17016574585635358, "loss": 0.0164, "grad_norm": 9.841991424560547, "learning_rate": 4.8696969696969705e-06, "num_tokens": 3012833.0, "completions/mean_length": 37.875, "completions/min_length": 36.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.875, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.941387414932251, "rewards/meter/std": 0.041819099336862564, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7620973587036133, "rewards/repeat_soft/std": 0.06565750390291214, "rewards/judge_quality/mean": 0.4024999737739563, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.5602039098739624, "rewards/total_composite/std": 0.040808357298374176, "reward": 0.5602039098739624, "reward_std": 0.040808361023664474, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04996867850422859, "sampling/sampling_logp_difference/max": 1.131651759147644, "sampling/importance_sampling_ratio/min": 0.3225001394748688, "sampling/importance_sampling_ratio/mean": 1.0246801376342773, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.24913730844855309, "clip_ratio/low_mean": 0.019243421033024788, "clip_ratio/low_min": 0.019243421033024788, "clip_ratio/high_mean": 0.026587442494928837, "clip_ratio/high_max": 0.026587442494928837, "clip_ratio/region_mean": 0.045830863527953625, "reward_total_mean": 0.5602039098739624, "reward_meter_mean": 0.941387414932251, "reward_meter_std": 0.041819099336862564, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7620973587036133, "reward_repeat_soft_std": 0.06565750390291214, "reward_judge_quality_mean": 0.4024999737739563, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.5602039098739624, "reward_total_composite_std": 0.040808357298374176} {"timestamp_utc": "2026-04-13T11:31:03Z", "mode": "train", "global_step": 1695, "epoch": 0.1702661978905073, "loss": 0.0639, "grad_norm": 13.700176239013672, "learning_rate": 4.866666666666667e-06, "num_tokens": 3014377.0, "completions/mean_length": 34.0, "completions/min_length": 32.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.8843082785606384, "rewards/meter/std": 0.2113535851240158, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7810590267181396, "rewards/repeat_soft/std": 0.09941042214632034, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5585750341415405, "rewards/total_composite/std": 0.04502662271261215, "reward": 0.5585750341415405, "reward_std": 0.04502662271261215, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08787927776575089, "sampling/sampling_logp_difference/max": 1.5108418464660645, "sampling/importance_sampling_ratio/min": 0.27825117111206055, "sampling/importance_sampling_ratio/mean": 1.0241531133651733, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.35129158198833466, "clip_ratio/low_mean": 0.02302631549537182, "clip_ratio/low_min": 0.02302631549537182, "clip_ratio/high_mean": 0.048711445881053805, "clip_ratio/high_max": 0.048711445881053805, "clip_ratio/region_mean": 0.07173776137642562, "reward_total_mean": 0.5585750341415405, "reward_meter_mean": 0.8843082785606384, "reward_meter_std": 0.2113535851240158, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7810590267181396, "reward_repeat_soft_std": 0.09941042214632034, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5585750341415405, "reward_total_composite_std": 0.04502662271261215} {"timestamp_utc": "2026-04-13T11:31:10Z", "mode": "train", "global_step": 1696, "epoch": 0.17036664992466097, "loss": 0.1091, "grad_norm": 7.173166275024414, "learning_rate": 4.863636363636364e-06, "num_tokens": 3016793.0, "completions/mean_length": 119.0, "completions/min_length": 98.0, "completions/max_length": 152.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 119.0, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 152.0, "rewards/meter/mean": 0.9621785283088684, "rewards/meter/std": 0.07766371965408325, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8270488977432251, "rewards/repeat_soft/std": 0.09540286660194397, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.586732029914856, "rewards/total_composite/std": 0.018162593245506287, "reward": 0.586732029914856, "reward_std": 0.018162589520215988, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09017331898212433, "sampling/sampling_logp_difference/max": 2.4064292907714844, "sampling/importance_sampling_ratio/min": 0.13300110399723053, "sampling/importance_sampling_ratio/mean": 1.0062044858932495, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.44541386142373085, "clip_ratio/low_mean": 0.015144678764045238, "clip_ratio/low_min": 0.015144678764045238, "clip_ratio/high_mean": 0.06057947035878897, "clip_ratio/high_max": 0.06057947035878897, "clip_ratio/region_mean": 0.0757241491228342, "reward_total_mean": 0.586732029914856, "reward_meter_mean": 0.9621785283088684, "reward_meter_std": 0.07766371965408325, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8270488977432251, "reward_repeat_soft_std": 0.09540286660194397, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.586732029914856, "reward_total_composite_std": 0.018162593245506287} {"timestamp_utc": "2026-04-13T11:31:17Z", "mode": "train", "global_step": 1697, "epoch": 0.17046710195881468, "loss": 0.002, "grad_norm": 3.858222246170044, "learning_rate": 4.8606060606060615e-06, "num_tokens": 3018934.0, "completions/mean_length": 104.625, "completions/min_length": 94.0, "completions/max_length": 112.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 104.625, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 112.0, "rewards/meter/mean": 0.9845653772354126, "rewards/meter/std": 0.011520404368638992, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.66322922706604, "rewards/repeat_soft/std": 0.10307466238737106, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.6074063777923584, "rewards/total_composite/std": 0.09574397653341293, "reward": 0.6074063777923584, "reward_std": 0.09574397653341293, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.059490155428647995, "sampling/sampling_logp_difference/max": 3.0207901000976562, "sampling/importance_sampling_ratio/min": 0.04876267537474632, "sampling/importance_sampling_ratio/mean": 1.0033087730407715, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.29337758757174015, "clip_ratio/low_mean": 0.04706494836136699, "clip_ratio/low_min": 0.04706494836136699, "clip_ratio/high_mean": 0.004807692486792803, "clip_ratio/high_max": 0.004807692486792803, "clip_ratio/region_mean": 0.05187264084815979, "reward_total_mean": 0.6074063777923584, "reward_meter_mean": 0.9845653772354126, "reward_meter_std": 0.011520404368638992, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.66322922706604, "reward_repeat_soft_std": 0.10307466238737106, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.6074063777923584, "reward_total_composite_std": 0.09574397653341293} {"timestamp_utc": "2026-04-13T11:31:23Z", "mode": "train", "global_step": 1698, "epoch": 0.17056755399296836, "loss": 0.0165, "grad_norm": 10.221336364746094, "learning_rate": 4.857575757575758e-06, "num_tokens": 3020568.0, "completions/mean_length": 35.25, "completions/min_length": 32.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.25, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9896067976951599, "rewards/meter/std": 0.0032773904968053102, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9455275535583496, "rewards/repeat_soft/std": 0.021588625386357307, "rewards/judge_quality/mean": 0.7112500071525574, "rewards/judge_quality/std": 0.24793073534965515, "rewards/total_composite/mean": 0.7994704246520996, "rewards/total_composite/std": 0.1603584885597229, "reward": 0.7994704246520996, "reward_std": 0.1603584885597229, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08277014642953873, "sampling/sampling_logp_difference/max": 1.3099803924560547, "sampling/importance_sampling_ratio/min": 0.26982536911964417, "sampling/importance_sampling_ratio/mean": 1.0136631727218628, "sampling/importance_sampling_ratio/max": 1.829031229019165, "entropy": 0.36017119884490967, "clip_ratio/low_mean": 0.03167505795136094, "clip_ratio/low_min": 0.03167505795136094, "clip_ratio/high_mean": 0.04956510406918824, "clip_ratio/high_max": 0.04956510406918824, "clip_ratio/region_mean": 0.08124016202054918, "reward_total_mean": 0.7994704246520996, "reward_meter_mean": 0.9896067976951599, "reward_meter_std": 0.0032773904968053102, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9455275535583496, "reward_repeat_soft_std": 0.021588625386357307, "reward_judge_quality_mean": 0.7112500071525574, "reward_judge_quality_std": 0.24793073534965515, "reward_total_composite_mean": 0.7994704246520996, "reward_total_composite_std": 0.1603584885597229} {"timestamp_utc": "2026-04-13T11:31:34Z", "mode": "train", "global_step": 1699, "epoch": 0.17066800602712204, "loss": -0.1534, "grad_norm": 1.69248366355896, "learning_rate": 4.854545454545455e-06, "num_tokens": 3022221.0, "completions/mean_length": 114.625, "completions/min_length": 42.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 57.857147216796875, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.8519529104232788, "rewards/meter/std": 0.3445809781551361, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9030239582061768, "rewards/repeat_soft/std": 0.06341450661420822, "rewards/judge_quality/mean": 0.3812500238418579, "rewards/judge_quality/std": 0.13452960550785065, "rewards/total_composite/mean": 0.5289825201034546, "rewards/total_composite/std": 0.21403877437114716, "reward": 0.5289825201034546, "reward_std": 0.21403877437114716, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11025475710630417, "sampling/sampling_logp_difference/max": 1.3818187713623047, "sampling/importance_sampling_ratio/min": 0.2511214017868042, "sampling/importance_sampling_ratio/mean": 1.0042784214019775, "sampling/importance_sampling_ratio/max": 1.9752295017242432, "entropy": 0.5506503693759441, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.1055633183568716, "clip_ratio/high_max": 0.1055633183568716, "clip_ratio/region_mean": 0.1055633183568716, "reward_total_mean": 0.5289825201034546, "reward_meter_mean": 0.8519529104232788, "reward_meter_std": 0.3445809781551361, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9030239582061768, "reward_repeat_soft_std": 0.06341450661420822, "reward_judge_quality_mean": 0.3812500238418579, "reward_judge_quality_std": 0.13452960550785065, "reward_total_composite_mean": 0.5289825201034546, "reward_total_composite_std": 0.21403877437114716} {"timestamp_utc": "2026-04-13T11:31:42Z", "mode": "train", "global_step": 1700, "epoch": 0.17076845806127575, "loss": 0.0074, "grad_norm": 5.452632904052734, "learning_rate": 4.851515151515152e-06, "num_tokens": 3024522.0, "completions/mean_length": 120.625, "completions/min_length": 106.0, "completions/max_length": 139.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 120.625, "completions/min_terminated_length": 106.0, "completions/max_terminated_length": 139.0, "rewards/meter/mean": 0.830030620098114, "rewards/meter/std": 0.2582976520061493, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7376050353050232, "rewards/repeat_soft/std": 0.09343899041414261, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.537239134311676, "rewards/total_composite/std": 0.06643049418926239, "reward": 0.537239134311676, "reward_std": 0.06643049418926239, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08235330134630203, "sampling/sampling_logp_difference/max": 1.233177661895752, "sampling/importance_sampling_ratio/min": 0.29136523604393005, "sampling/importance_sampling_ratio/mean": 1.004504919052124, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.43990712240338326, "clip_ratio/low_mean": 0.01977576781064272, "clip_ratio/low_min": 0.01977576781064272, "clip_ratio/high_mean": 0.06406635232269764, "clip_ratio/high_max": 0.06406635232269764, "clip_ratio/region_mean": 0.08384212013334036, "reward_total_mean": 0.537239134311676, "reward_meter_mean": 0.830030620098114, "reward_meter_std": 0.2582976520061493, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7376050353050232, "reward_repeat_soft_std": 0.09343899041414261, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.537239134311676, "reward_total_composite_std": 0.06643049418926239} {"timestamp_utc": "2026-04-13T11:32:34Z", "mode": "eval", "global_step": 1700, "epoch": 0.17076845806127575, "eval_loss": NaN, "eval_runtime": 51.766, "eval_samples_per_second": 1.545, "eval_steps_per_second": 0.193, "eval_num_tokens": 3024522.0, "eval_completions/mean_length": 96.95, "eval_completions/min_length": 34.7, "eval_completions/max_length": 210.6, "eval_completions/clipped_ratio": 0.0375, "eval_completions/mean_terminated_length": 81.07321472167969, "eval_completions/min_terminated_length": 34.7, "eval_completions/max_terminated_length": 145.1, "eval_rewards/meter/mean": 0.7788607656955719, "eval_rewards/meter/std": 0.2661908954381943, "eval_rewards/count_adherence/mean": 0.951874989271164, "eval_rewards/count_adherence/std": 0.08645724020898342, "eval_rewards/hard_gate/mean": 0.9375, "eval_rewards/hard_gate/std": 0.09804592728614807, "eval_rewards/repeat_soft/mean": 0.8310980260372162, "eval_rewards/repeat_soft/std": 0.11172637566924096, "eval_rewards/judge_quality/mean": 0.42000000476837157, "eval_rewards/judge_quality/std": 0.1566985785961151, "eval_rewards/total_composite/mean": 0.5148498684167861, "eval_rewards/total_composite/std": 0.1532291118055582, "eval_reward": 0.5148498684167861, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.04819921813905239, "eval_sampling/sampling_logp_difference/max": 1.0375221252441407, "eval_sampling/importance_sampling_ratio/min": 0.36269562840461733, "eval_sampling/importance_sampling_ratio/mean": 1.0100383281707763, "eval_sampling/importance_sampling_ratio/max": 1.4731812477111816, "eval_entropy": 0.49511538743972777, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5148498684167861, "eval_reward_meter_mean": 0.7788607656955719, "eval_reward_meter_std": 0.2661908954381943, "eval_reward_count_adherence_mean": 0.951874989271164, "eval_reward_count_adherence_std": 0.08645724020898342, "eval_reward_hard_gate_mean": 0.9375, "eval_reward_hard_gate_std": 0.09804592728614807, "eval_reward_repeat_soft_mean": 0.8310980260372162, "eval_reward_repeat_soft_std": 0.11172637566924096, "eval_reward_judge_quality_mean": 0.42000000476837157, "eval_reward_judge_quality_std": 0.1566985785961151, "eval_reward_total_composite_mean": 0.5148498684167861, "eval_reward_total_composite_std": 0.1532291118055582} {"timestamp_utc": "2026-04-13T11:32:44Z", "mode": "train", "global_step": 1701, "epoch": 0.17086891009542943, "loss": -0.0625, "grad_norm": 6.982199668884277, "learning_rate": 4.848484848484849e-06, "num_tokens": 3026667.0, "completions/mean_length": 100.125, "completions/min_length": 73.0, "completions/max_length": 115.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.125, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 115.0, "rewards/meter/mean": 0.9745658040046692, "rewards/meter/std": 0.030551763251423836, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7237980365753174, "rewards/repeat_soft/std": 0.06728465110063553, "rewards/judge_quality/mean": 0.4025000035762787, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.56345534324646, "rewards/total_composite/std": 0.04501413553953171, "reward": 0.56345534324646, "reward_std": 0.04501413553953171, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07348770648241043, "sampling/sampling_logp_difference/max": 1.2052745819091797, "sampling/importance_sampling_ratio/min": 0.2996097207069397, "sampling/importance_sampling_ratio/mean": 1.001634955406189, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.450840562582016, "clip_ratio/low_mean": 0.012467292603105307, "clip_ratio/low_min": 0.012467292603105307, "clip_ratio/high_mean": 0.05064171692356467, "clip_ratio/high_max": 0.05064171692356467, "clip_ratio/region_mean": 0.06310900952666998, "reward_total_mean": 0.56345534324646, "reward_meter_mean": 0.9745658040046692, "reward_meter_std": 0.030551763251423836, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7237980365753174, "reward_repeat_soft_std": 0.06728465110063553, "reward_judge_quality_mean": 0.4025000035762787, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.56345534324646, "reward_total_composite_std": 0.04501413553953171} {"timestamp_utc": "2026-04-13T11:32:50Z", "mode": "train", "global_step": 1702, "epoch": 0.17096936212958314, "loss": 0.0219, "grad_norm": 5.328232288360596, "learning_rate": 4.845454545454546e-06, "num_tokens": 3028905.0, "completions/mean_length": 110.75, "completions/min_length": 93.0, "completions/max_length": 121.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 110.75, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.9815160632133484, "rewards/meter/std": 0.010495602153241634, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7253273725509644, "rewards/repeat_soft/std": 0.08580384403467178, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.5415375232696533, "rewards/total_composite/std": 0.07670989632606506, "reward": 0.5415375232696533, "reward_std": 0.07670990377664566, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07422178238630295, "sampling/sampling_logp_difference/max": 1.5198631286621094, "sampling/importance_sampling_ratio/min": 0.28725555539131165, "sampling/importance_sampling_ratio/mean": 1.0084192752838135, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4771435223519802, "clip_ratio/low_mean": 0.009179437882266939, "clip_ratio/low_min": 0.009179437882266939, "clip_ratio/high_mean": 0.06365141971036792, "clip_ratio/high_max": 0.06365141971036792, "clip_ratio/region_mean": 0.07283085759263486, "reward_total_mean": 0.5415375232696533, "reward_meter_mean": 0.9815160632133484, "reward_meter_std": 0.010495602153241634, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7253273725509644, "reward_repeat_soft_std": 0.08580384403467178, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.5415375232696533, "reward_total_composite_std": 0.07670989632606506} {"timestamp_utc": "2026-04-13T11:32:58Z", "mode": "train", "global_step": 1703, "epoch": 0.17106981416373682, "loss": 0.0537, "grad_norm": 7.986655235290527, "learning_rate": 4.842424242424243e-06, "num_tokens": 3031039.0, "completions/mean_length": 91.75, "completions/min_length": 77.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.75, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.8958242535591125, "rewards/meter/std": 0.2371399998664856, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8050409555435181, "rewards/repeat_soft/std": 0.049173615872859955, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.10966669768095016, "rewards/total_composite/mean": 0.517031192779541, "rewards/total_composite/std": 0.0822354108095169, "reward": 0.517031192779541, "reward_std": 0.0822354108095169, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09671947360038757, "sampling/sampling_logp_difference/max": 2.264767646789551, "sampling/importance_sampling_ratio/min": 0.10385415703058243, "sampling/importance_sampling_ratio/mean": 1.0112695693969727, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5040853545069695, "clip_ratio/low_mean": 0.043156355153769255, "clip_ratio/low_min": 0.043156355153769255, "clip_ratio/high_mean": 0.044979365542531013, "clip_ratio/high_max": 0.044979365542531013, "clip_ratio/region_mean": 0.08813572069630027, "reward_total_mean": 0.517031192779541, "reward_meter_mean": 0.8958242535591125, "reward_meter_std": 0.2371399998664856, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8050409555435181, "reward_repeat_soft_std": 0.049173615872859955, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.10966669768095016, "reward_total_composite_mean": 0.517031192779541, "reward_total_composite_std": 0.0822354108095169} {"timestamp_utc": "2026-04-13T11:33:04Z", "mode": "train", "global_step": 1704, "epoch": 0.1711702661978905, "loss": 0.0574, "grad_norm": 12.21061897277832, "learning_rate": 4.83939393939394e-06, "num_tokens": 3032720.0, "completions/mean_length": 41.125, "completions/min_length": 31.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.125, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.7094885110855103, "rewards/meter/std": 0.27525606751441956, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9190400242805481, "rewards/repeat_soft/std": 0.03655832260847092, "rewards/judge_quality/mean": 0.5387499928474426, "rewards/judge_quality/std": 0.1888640969991684, "rewards/total_composite/mean": 0.5927785634994507, "rewards/total_composite/std": 0.13393543660640717, "reward": 0.5927785634994507, "reward_std": 0.13393543660640717, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11570199579000473, "sampling/sampling_logp_difference/max": 2.3757834434509277, "sampling/importance_sampling_ratio/min": 0.09294164925813675, "sampling/importance_sampling_ratio/mean": 1.0100421905517578, "sampling/importance_sampling_ratio/max": 1.8284733295440674, "entropy": 0.6329236328601837, "clip_ratio/low_mean": 0.04719053162261844, "clip_ratio/low_min": 0.04719053162261844, "clip_ratio/high_mean": 0.04858295898884535, "clip_ratio/high_max": 0.04858295898884535, "clip_ratio/region_mean": 0.09577349061146379, "reward_total_mean": 0.5927785634994507, "reward_meter_mean": 0.7094885110855103, "reward_meter_std": 0.27525606751441956, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9190400242805481, "reward_repeat_soft_std": 0.03655832260847092, "reward_judge_quality_mean": 0.5387499928474426, "reward_judge_quality_std": 0.1888640969991684, "reward_total_composite_mean": 0.5927785634994507, "reward_total_composite_std": 0.13393543660640717} {"timestamp_utc": "2026-04-13T11:33:10Z", "mode": "train", "global_step": 1705, "epoch": 0.1712707182320442, "loss": -0.0042, "grad_norm": 19.694028854370117, "learning_rate": 4.836363636363637e-06, "num_tokens": 3034208.0, "completions/mean_length": 24.0, "completions/min_length": 20.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.0, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.8004680871963501, "rewards/meter/std": 0.2228754162788391, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9544928669929504, "rewards/repeat_soft/std": 0.015905901789665222, "rewards/judge_quality/mean": 0.3387500047683716, "rewards/judge_quality/std": 0.13292719423770905, "rewards/total_composite/mean": 0.5244325995445251, "rewards/total_composite/std": 0.09833145141601562, "reward": 0.5244325995445251, "reward_std": 0.09833145141601562, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11064853519201279, "sampling/sampling_logp_difference/max": 1.2437875270843506, "sampling/importance_sampling_ratio/min": 0.28829023241996765, "sampling/importance_sampling_ratio/mean": 0.9975305795669556, "sampling/importance_sampling_ratio/max": 1.6610095500946045, "entropy": 0.86296296864748, "clip_ratio/low_mean": 0.041220239363610744, "clip_ratio/low_min": 0.041220239363610744, "clip_ratio/high_mean": 0.04556785896420479, "clip_ratio/high_max": 0.04556785896420479, "clip_ratio/region_mean": 0.08678809832781553, "reward_total_mean": 0.5244325995445251, "reward_meter_mean": 0.8004680871963501, "reward_meter_std": 0.2228754162788391, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9544928669929504, "reward_repeat_soft_std": 0.015905901789665222, "reward_judge_quality_mean": 0.3387500047683716, "reward_judge_quality_std": 0.13292719423770905, "reward_total_composite_mean": 0.5244325995445251, "reward_total_composite_std": 0.09833145141601562} {"timestamp_utc": "2026-04-13T11:33:17Z", "mode": "train", "global_step": 1706, "epoch": 0.1713711702661979, "loss": 0.0292, "grad_norm": 12.84737777709961, "learning_rate": 4.833333333333333e-06, "num_tokens": 3035743.0, "completions/mean_length": 43.875, "completions/min_length": 38.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.875, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.7723362445831299, "rewards/meter/std": 0.2725357413291931, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9816555976867676, "rewards/repeat_soft/std": 0.04615623131394386, "rewards/judge_quality/mean": 0.5575000047683716, "rewards/judge_quality/std": 0.19955310225486755, "rewards/total_composite/mean": 0.6426393985748291, "rewards/total_composite/std": 0.1737646460533142, "reward": 0.6426393985748291, "reward_std": 0.1737646460533142, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1071280688047409, "sampling/sampling_logp_difference/max": 1.0546283721923828, "sampling/importance_sampling_ratio/min": 0.3483218550682068, "sampling/importance_sampling_ratio/mean": 1.0121512413024902, "sampling/importance_sampling_ratio/max": 1.9893280267715454, "entropy": 0.6153622791171074, "clip_ratio/low_mean": 0.100666300393641, "clip_ratio/low_min": 0.100666300393641, "clip_ratio/high_mean": 0.03223467618227005, "clip_ratio/high_max": 0.03223467618227005, "clip_ratio/region_mean": 0.13290097657591105, "reward_total_mean": 0.6426393985748291, "reward_meter_mean": 0.7723362445831299, "reward_meter_std": 0.2725357413291931, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9816555976867676, "reward_repeat_soft_std": 0.04615623131394386, "reward_judge_quality_mean": 0.5575000047683716, "reward_judge_quality_std": 0.19955310225486755, "reward_total_composite_mean": 0.6426393985748291, "reward_total_composite_std": 0.1737646460533142} {"timestamp_utc": "2026-04-13T11:33:24Z", "mode": "train", "global_step": 1707, "epoch": 0.1714716223003516, "loss": -0.0433, "grad_norm": 12.744470596313477, "learning_rate": 4.830303030303031e-06, "num_tokens": 3037470.0, "completions/mean_length": 53.875, "completions/min_length": 45.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.875, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9339768886566162, "rewards/meter/std": 0.12653842568397522, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9758355021476746, "rewards/repeat_soft/std": 0.037519022822380066, "rewards/judge_quality/mean": 0.5274999737739563, "rewards/judge_quality/std": 0.18873640894889832, "rewards/total_composite/mean": 0.6698579788208008, "rewards/total_composite/std": 0.13340696692466736, "reward": 0.6698579788208008, "reward_std": 0.13340695202350616, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12084158509969711, "sampling/sampling_logp_difference/max": 1.859917163848877, "sampling/importance_sampling_ratio/min": 0.15568551421165466, "sampling/importance_sampling_ratio/mean": 1.0078500509262085, "sampling/importance_sampling_ratio/max": 1.920824646949768, "entropy": 0.8653598427772522, "clip_ratio/low_mean": 0.07423791708424687, "clip_ratio/low_min": 0.07423791708424687, "clip_ratio/high_mean": 0.031413814052939415, "clip_ratio/high_max": 0.031413814052939415, "clip_ratio/region_mean": 0.10565173113718629, "reward_total_mean": 0.6698579788208008, "reward_meter_mean": 0.9339768886566162, "reward_meter_std": 0.12653842568397522, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9758355021476746, "reward_repeat_soft_std": 0.037519022822380066, "reward_judge_quality_mean": 0.5274999737739563, "reward_judge_quality_std": 0.18873640894889832, "reward_total_composite_mean": 0.6698579788208008, "reward_total_composite_std": 0.13340696692466736} {"timestamp_utc": "2026-04-13T11:33:30Z", "mode": "train", "global_step": 1708, "epoch": 0.17157207433450528, "loss": -0.0337, "grad_norm": 8.233144760131836, "learning_rate": 4.827272727272728e-06, "num_tokens": 3039121.0, "completions/mean_length": 37.375, "completions/min_length": 31.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.375, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.45985865592956543, "rewards/meter/std": 0.42175498604774475, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8944973945617676, "rewards/repeat_soft/std": 0.04428502172231674, "rewards/judge_quality/mean": 0.4724999964237213, "rewards/judge_quality/std": 0.1011011004447937, "rewards/total_composite/mean": 0.4863719940185547, "rewards/total_composite/std": 0.15772169828414917, "reward": 0.4863719940185547, "reward_std": 0.15772169828414917, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09548614919185638, "sampling/sampling_logp_difference/max": 1.1301515102386475, "sampling/importance_sampling_ratio/min": 0.3229843080043793, "sampling/importance_sampling_ratio/mean": 1.0027332305908203, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.586298007518053, "clip_ratio/low_mean": 0.05920380586758256, "clip_ratio/low_min": 0.05920380586758256, "clip_ratio/high_mean": 0.037892288994044065, "clip_ratio/high_max": 0.037892288994044065, "clip_ratio/region_mean": 0.09709609486162663, "reward_total_mean": 0.4863719940185547, "reward_meter_mean": 0.45985865592956543, "reward_meter_std": 0.42175498604774475, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8944973945617676, "reward_repeat_soft_std": 0.04428502172231674, "reward_judge_quality_mean": 0.4724999964237213, "reward_judge_quality_std": 0.1011011004447937, "reward_total_composite_mean": 0.4863719940185547, "reward_total_composite_std": 0.15772169828414917} {"timestamp_utc": "2026-04-13T11:33:37Z", "mode": "train", "global_step": 1709, "epoch": 0.17167252636865896, "loss": 0.0363, "grad_norm": 9.889695167541504, "learning_rate": 4.824242424242424e-06, "num_tokens": 3040825.0, "completions/mean_length": 47.0, "completions/min_length": 40.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.0, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9627807140350342, "rewards/meter/std": 0.07612529397010803, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9161338806152344, "rewards/repeat_soft/std": 0.044543150812387466, "rewards/judge_quality/mean": 0.47999998927116394, "rewards/judge_quality/std": 0.10993505269289017, "rewards/total_composite/mean": 0.6330803632736206, "rewards/total_composite/std": 0.033940430730581284, "reward": 0.6330803632736206, "reward_std": 0.03394043818116188, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10801807045936584, "sampling/sampling_logp_difference/max": 1.71560537815094, "sampling/importance_sampling_ratio/min": 0.17985482513904572, "sampling/importance_sampling_ratio/mean": 0.9994314908981323, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5231927931308746, "clip_ratio/low_mean": 0.05567022366449237, "clip_ratio/low_min": 0.05567022366449237, "clip_ratio/high_mean": 0.0384084889665246, "clip_ratio/high_max": 0.0384084889665246, "clip_ratio/region_mean": 0.09407871263101697, "reward_total_mean": 0.6330803632736206, "reward_meter_mean": 0.9627807140350342, "reward_meter_std": 0.07612529397010803, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9161338806152344, "reward_repeat_soft_std": 0.044543150812387466, "reward_judge_quality_mean": 0.47999998927116394, "reward_judge_quality_std": 0.10993505269289017, "reward_total_composite_mean": 0.6330803632736206, "reward_total_composite_std": 0.033940430730581284} {"timestamp_utc": "2026-04-13T11:33:45Z", "mode": "train", "global_step": 1710, "epoch": 0.17177297840281266, "loss": 0.1373, "grad_norm": 7.542257785797119, "learning_rate": 4.8212121212121215e-06, "num_tokens": 3042836.0, "completions/mean_length": 83.375, "completions/min_length": 66.0, "completions/max_length": 126.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 83.375, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 126.0, "rewards/meter/mean": 0.3569932281970978, "rewards/meter/std": 0.2999321222305298, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8213039636611938, "rewards/repeat_soft/std": 0.10220811516046524, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.41422730684280396, "rewards/total_composite/std": 0.09932152181863785, "reward": 0.41422730684280396, "reward_std": 0.09932152181863785, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09379032999277115, "sampling/sampling_logp_difference/max": 1.533194899559021, "sampling/importance_sampling_ratio/min": 0.21584497392177582, "sampling/importance_sampling_ratio/mean": 1.0104491710662842, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5609025619924068, "clip_ratio/low_mean": 0.04872157471254468, "clip_ratio/low_min": 0.04872157471254468, "clip_ratio/high_mean": 0.04706374555826187, "clip_ratio/high_max": 0.04706374555826187, "clip_ratio/region_mean": 0.09578532027080655, "reward_total_mean": 0.41422730684280396, "reward_meter_mean": 0.3569932281970978, "reward_meter_std": 0.2999321222305298, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8213039636611938, "reward_repeat_soft_std": 0.10220811516046524, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.41422730684280396, "reward_total_composite_std": 0.09932152181863785} {"timestamp_utc": "2026-04-13T11:33:56Z", "mode": "train", "global_step": 1711, "epoch": 0.17187343043696635, "loss": -0.2413, "grad_norm": 2.1527371406555176, "learning_rate": 4.818181818181819e-06, "num_tokens": 3045448.0, "completions/mean_length": 205.5, "completions/min_length": 128.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 161.71429443359375, "completions/min_terminated_length": 128.0, "completions/max_terminated_length": 181.0, "rewards/meter/mean": 0.8527488112449646, "rewards/meter/std": 0.23914027214050293, "rewards/count_adherence/mean": 0.8958333730697632, "rewards/count_adherence/std": 0.23464766144752502, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.853449285030365, "rewards/repeat_soft/std": 0.08920948952436447, "rewards/judge_quality/mean": 0.24250000715255737, "rewards/judge_quality/std": 0.12947696447372437, "rewards/total_composite/mean": 0.4225049614906311, "rewards/total_composite/std": 0.18557484447956085, "reward": 0.4225049614906311, "reward_std": 0.18557484447956085, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08878102898597717, "sampling/sampling_logp_difference/max": 3.4642715454101562, "sampling/importance_sampling_ratio/min": 0.03129579499363899, "sampling/importance_sampling_ratio/mean": 1.012242317199707, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5966197475790977, "clip_ratio/low_mean": 0.007867133244872093, "clip_ratio/low_min": 0.007867133244872093, "clip_ratio/high_mean": 0.07043249066919088, "clip_ratio/high_max": 0.07043249066919088, "clip_ratio/region_mean": 0.07829962391406298, "reward_total_mean": 0.4225049614906311, "reward_meter_mean": 0.8527488112449646, "reward_meter_std": 0.23914027214050293, "reward_count_adherence_mean": 0.8958333730697632, "reward_count_adherence_std": 0.23464766144752502, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.853449285030365, "reward_repeat_soft_std": 0.08920948952436447, "reward_judge_quality_mean": 0.24250000715255737, "reward_judge_quality_std": 0.12947696447372437, "reward_total_composite_mean": 0.4225049614906311, "reward_total_composite_std": 0.18557484447956085} {"timestamp_utc": "2026-04-13T11:34:08Z", "mode": "train", "global_step": 1712, "epoch": 0.17197388247112003, "loss": -0.1346, "grad_norm": 3.0581750869750977, "learning_rate": 4.815151515151515e-06, "num_tokens": 3047159.0, "completions/mean_length": 113.875, "completions/min_length": 53.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 57.000003814697266, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.7386385202407837, "rewards/meter/std": 0.4336848556995392, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.983648419380188, "rewards/repeat_soft/std": 0.016031138598918915, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.13617216050624847, "rewards/total_composite/mean": 0.5108710527420044, "rewards/total_composite/std": 0.22471150755882263, "reward": 0.5108710527420044, "reward_std": 0.22471149265766144, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11820080131292343, "sampling/sampling_logp_difference/max": 2.4018521308898926, "sampling/importance_sampling_ratio/min": 0.09055008739233017, "sampling/importance_sampling_ratio/mean": 1.0040462017059326, "sampling/importance_sampling_ratio/max": 1.9066431522369385, "entropy": 0.8000173345208168, "clip_ratio/low_mean": 0.004545454401522875, "clip_ratio/low_min": 0.004545454401522875, "clip_ratio/high_mean": 0.08814913779497147, "clip_ratio/high_max": 0.08814913779497147, "clip_ratio/region_mean": 0.09269459219649434, "reward_total_mean": 0.5108710527420044, "reward_meter_mean": 0.7386385202407837, "reward_meter_std": 0.4336848556995392, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.983648419380188, "reward_repeat_soft_std": 0.016031138598918915, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.13617216050624847, "reward_total_composite_mean": 0.5108710527420044, "reward_total_composite_std": 0.22471150755882263} {"timestamp_utc": "2026-04-13T11:34:15Z", "mode": "train", "global_step": 1713, "epoch": 0.17207433450527373, "loss": 0.0401, "grad_norm": 8.389473915100098, "learning_rate": 4.8121212121212125e-06, "num_tokens": 3049314.0, "completions/mean_length": 110.375, "completions/min_length": 82.0, "completions/max_length": 123.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 110.375, "completions/min_terminated_length": 82.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.802635669708252, "rewards/meter/std": 0.19809673726558685, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9171019792556763, "rewards/repeat_soft/std": 0.03333454951643944, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5566848516464233, "rewards/total_composite/std": 0.05519948527216911, "reward": 0.5566848516464233, "reward_std": 0.05519949272274971, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10923045128583908, "sampling/sampling_logp_difference/max": 2.0601401329040527, "sampling/importance_sampling_ratio/min": 0.12743611633777618, "sampling/importance_sampling_ratio/mean": 1.003833532333374, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5601277053356171, "clip_ratio/low_mean": 0.022562035359442234, "clip_ratio/low_min": 0.022562035359442234, "clip_ratio/high_mean": 0.05748207215219736, "clip_ratio/high_max": 0.05748207215219736, "clip_ratio/region_mean": 0.0800441075116396, "reward_total_mean": 0.5566848516464233, "reward_meter_mean": 0.802635669708252, "reward_meter_std": 0.19809673726558685, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9171019792556763, "reward_repeat_soft_std": 0.03333454951643944, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5566848516464233, "reward_total_composite_std": 0.05519948527216911} {"timestamp_utc": "2026-04-13T11:34:21Z", "mode": "train", "global_step": 1714, "epoch": 0.17217478653942742, "loss": 0.0001, "grad_norm": 9.25256633758545, "learning_rate": 4.80909090909091e-06, "num_tokens": 3050797.0, "completions/mean_length": 36.375, "completions/min_length": 34.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.375, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.4204741418361664, "rewards/meter/std": 0.336512953042984, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8884314894676208, "rewards/repeat_soft/std": 0.050489891320466995, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.4530245363712311, "rewards/total_composite/std": 0.09045282006263733, "reward": 0.4530245363712311, "reward_std": 0.09045282006263733, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0816180482506752, "sampling/sampling_logp_difference/max": 2.278085231781006, "sampling/importance_sampling_ratio/min": 0.10248024016618729, "sampling/importance_sampling_ratio/mean": 0.9892809987068176, "sampling/importance_sampling_ratio/max": 1.8046834468841553, "entropy": 0.44643400982022285, "clip_ratio/low_mean": 0.03467508847825229, "clip_ratio/low_min": 0.03467508847825229, "clip_ratio/high_mean": 0.0642518401145935, "clip_ratio/high_max": 0.0642518401145935, "clip_ratio/region_mean": 0.0989269285928458, "reward_total_mean": 0.4530245363712311, "reward_meter_mean": 0.4204741418361664, "reward_meter_std": 0.336512953042984, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8884314894676208, "reward_repeat_soft_std": 0.050489891320466995, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.4530245363712311, "reward_total_composite_std": 0.09045282006263733} {"timestamp_utc": "2026-04-13T11:34:27Z", "mode": "train", "global_step": 1715, "epoch": 0.17227523857358112, "loss": -0.0584, "grad_norm": 9.323890686035156, "learning_rate": 4.806060606060606e-06, "num_tokens": 3052563.0, "completions/mean_length": 50.75, "completions/min_length": 38.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.75, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9446964263916016, "rewards/meter/std": 0.09015833586454391, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8592045903205872, "rewards/repeat_soft/std": 0.026658890768885612, "rewards/judge_quality/mean": 0.6225000023841858, "rewards/judge_quality/std": 0.24656209349632263, "rewards/total_composite/mean": 0.7154438495635986, "rewards/total_composite/std": 0.1673603504896164, "reward": 0.7154438495635986, "reward_std": 0.1673603504896164, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07247844338417053, "sampling/sampling_logp_difference/max": 0.961169958114624, "sampling/importance_sampling_ratio/min": 0.38244518637657166, "sampling/importance_sampling_ratio/mean": 1.01378333568573, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4564861096441746, "clip_ratio/low_mean": 0.05797531921416521, "clip_ratio/low_min": 0.05797531921416521, "clip_ratio/high_mean": 0.04102809727191925, "clip_ratio/high_max": 0.04102809727191925, "clip_ratio/region_mean": 0.09900341648608446, "reward_total_mean": 0.7154438495635986, "reward_meter_mean": 0.9446964263916016, "reward_meter_std": 0.09015833586454391, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8592045903205872, "reward_repeat_soft_std": 0.026658890768885612, "reward_judge_quality_mean": 0.6225000023841858, "reward_judge_quality_std": 0.24656209349632263, "reward_total_composite_mean": 0.7154438495635986, "reward_total_composite_std": 0.1673603504896164} {"timestamp_utc": "2026-04-13T11:34:34Z", "mode": "train", "global_step": 1716, "epoch": 0.1723756906077348, "loss": -0.0046, "grad_norm": 13.035734176635742, "learning_rate": 4.803030303030303e-06, "num_tokens": 3054350.0, "completions/mean_length": 52.375, "completions/min_length": 42.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.375, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.8809199929237366, "rewards/meter/std": 0.20748481154441833, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9217410683631897, "rewards/repeat_soft/std": 0.033956173807382584, "rewards/judge_quality/mean": 0.6775000095367432, "rewards/judge_quality/std": 0.25949129462242126, "rewards/total_composite/mean": 0.7109391093254089, "rewards/total_composite/std": 0.15203580260276794, "reward": 0.7109391093254089, "reward_std": 0.15203580260276794, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13106243312358856, "sampling/sampling_logp_difference/max": 2.2116916179656982, "sampling/importance_sampling_ratio/min": 0.1095152348279953, "sampling/importance_sampling_ratio/mean": 1.009244680404663, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6339414864778519, "clip_ratio/low_mean": 0.06317034922540188, "clip_ratio/low_min": 0.06317034922540188, "clip_ratio/high_mean": 0.04133064579218626, "clip_ratio/high_max": 0.04133064579218626, "clip_ratio/region_mean": 0.10450099501758814, "reward_total_mean": 0.7109391093254089, "reward_meter_mean": 0.8809199929237366, "reward_meter_std": 0.20748481154441833, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9217410683631897, "reward_repeat_soft_std": 0.033956173807382584, "reward_judge_quality_mean": 0.6775000095367432, "reward_judge_quality_std": 0.25949129462242126, "reward_total_composite_mean": 0.7109391093254089, "reward_total_composite_std": 0.15203580260276794} {"timestamp_utc": "2026-04-13T11:34:40Z", "mode": "train", "global_step": 1717, "epoch": 0.17247614264188849, "loss": 0.0767, "grad_norm": 14.608574867248535, "learning_rate": 4.800000000000001e-06, "num_tokens": 3056031.0, "completions/mean_length": 40.125, "completions/min_length": 32.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.125, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.4885815680027008, "rewards/meter/std": 0.3847001791000366, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.997380793094635, "rewards/repeat_soft/std": 0.003361919429153204, "rewards/judge_quality/mean": 0.7987500429153442, "rewards/judge_quality/std": 0.22465451061725616, "rewards/total_composite/mean": 0.6047747731208801, "rewards/total_composite/std": 0.22018471360206604, "reward": 0.6047747731208801, "reward_std": 0.22018472850322723, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12028294056653976, "sampling/sampling_logp_difference/max": 1.6750097274780273, "sampling/importance_sampling_ratio/min": 0.18730635941028595, "sampling/importance_sampling_ratio/mean": 0.9817869663238525, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4833027049899101, "clip_ratio/low_mean": 0.06514029623940587, "clip_ratio/low_min": 0.06514029623940587, "clip_ratio/high_mean": 0.037118902895599604, "clip_ratio/high_max": 0.037118902895599604, "clip_ratio/region_mean": 0.10225919913500547, "reward_total_mean": 0.6047747731208801, "reward_meter_mean": 0.4885815680027008, "reward_meter_std": 0.3847001791000366, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.997380793094635, "reward_repeat_soft_std": 0.003361919429153204, "reward_judge_quality_mean": 0.7987500429153442, "reward_judge_quality_std": 0.22465451061725616, "reward_total_composite_mean": 0.6047747731208801, "reward_total_composite_std": 0.22018471360206604} {"timestamp_utc": "2026-04-13T11:34:46Z", "mode": "train", "global_step": 1718, "epoch": 0.1725765946760422, "loss": 0.0068, "grad_norm": 14.311399459838867, "learning_rate": 4.796969696969697e-06, "num_tokens": 3057577.0, "completions/mean_length": 21.25, "completions/min_length": 20.0, "completions/max_length": 22.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.25, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 22.0, "rewards/meter/mean": 0.9603111147880554, "rewards/meter/std": 0.06461824476718903, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.6089694499969482, "rewards/total_composite/std": 0.0203370600938797, "reward": 0.6089694499969482, "reward_std": 0.020337076857686043, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07003674656152725, "sampling/sampling_logp_difference/max": 0.9432845115661621, "sampling/importance_sampling_ratio/min": 0.3893469274044037, "sampling/importance_sampling_ratio/mean": 0.9982831478118896, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.36864886805415154, "clip_ratio/low_mean": 0.030113637447357178, "clip_ratio/low_min": 0.030113637447357178, "clip_ratio/high_mean": 0.0525162355042994, "clip_ratio/high_max": 0.0525162355042994, "clip_ratio/region_mean": 0.08262987295165658, "reward_total_mean": 0.6089694499969482, "reward_meter_mean": 0.9603111147880554, "reward_meter_std": 0.06461824476718903, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.6089694499969482, "reward_total_composite_std": 0.0203370600938797} {"timestamp_utc": "2026-04-13T11:34:53Z", "mode": "train", "global_step": 1719, "epoch": 0.17267704671019588, "loss": -0.0779, "grad_norm": 15.311482429504395, "learning_rate": 4.793939393939394e-06, "num_tokens": 3059250.0, "completions/mean_length": 46.125, "completions/min_length": 38.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.125, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.44882452487945557, "rewards/meter/std": 0.4111330211162567, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8626612424850464, "rewards/repeat_soft/std": 0.13374517858028412, "rewards/judge_quality/mean": 0.7325000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.5474143624305725, "rewards/total_composite/std": 0.22817520797252655, "reward": 0.5474143624305725, "reward_std": 0.22817519307136536, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11525116860866547, "sampling/sampling_logp_difference/max": 2.725635290145874, "sampling/importance_sampling_ratio/min": 0.06550458073616028, "sampling/importance_sampling_ratio/mean": 1.0321239233016968, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5838736332952976, "clip_ratio/low_mean": 0.06504121143370867, "clip_ratio/low_min": 0.06504121143370867, "clip_ratio/high_mean": 0.020101881120353937, "clip_ratio/high_max": 0.020101881120353937, "clip_ratio/region_mean": 0.0851430925540626, "reward_total_mean": 0.5474143624305725, "reward_meter_mean": 0.44882452487945557, "reward_meter_std": 0.4111330211162567, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8626612424850464, "reward_repeat_soft_std": 0.13374517858028412, "reward_judge_quality_mean": 0.7325000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.5474143624305725, "reward_total_composite_std": 0.22817520797252655} {"timestamp_utc": "2026-04-13T11:35:01Z", "mode": "train", "global_step": 1720, "epoch": 0.17277749874434958, "loss": 0.0429, "grad_norm": 6.603409290313721, "learning_rate": 4.790909090909091e-06, "num_tokens": 3061572.0, "completions/mean_length": 110.25, "completions/min_length": 99.0, "completions/max_length": 125.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 110.25, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.9676520228385925, "rewards/meter/std": 0.032065149396657944, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7556529641151428, "rewards/repeat_soft/std": 0.08880481123924255, "rewards/judge_quality/mean": 0.5325000286102295, "rewards/judge_quality/std": 0.15526477992534637, "rewards/total_composite/mean": 0.6472546458244324, "rewards/total_composite/std": 0.10062187165021896, "reward": 0.6472546458244324, "reward_std": 0.10062186419963837, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09675262123346329, "sampling/sampling_logp_difference/max": 1.920860767364502, "sampling/importance_sampling_ratio/min": 0.14648081362247467, "sampling/importance_sampling_ratio/mean": 1.0061516761779785, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5555191040039062, "clip_ratio/low_mean": 0.053047848865389824, "clip_ratio/low_min": 0.053047848865389824, "clip_ratio/high_mean": 0.03863520547747612, "clip_ratio/high_max": 0.03863520547747612, "clip_ratio/region_mean": 0.09168305434286594, "reward_total_mean": 0.6472546458244324, "reward_meter_mean": 0.9676520228385925, "reward_meter_std": 0.032065149396657944, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7556529641151428, "reward_repeat_soft_std": 0.08880481123924255, "reward_judge_quality_mean": 0.5325000286102295, "reward_judge_quality_std": 0.15526477992534637, "reward_total_composite_mean": 0.6472546458244324, "reward_total_composite_std": 0.10062187165021896} {"timestamp_utc": "2026-04-13T11:35:08Z", "mode": "train", "global_step": 1721, "epoch": 0.17287795077850326, "loss": 0.0166, "grad_norm": 12.26235294342041, "learning_rate": 4.787878787878788e-06, "num_tokens": 3063239.0, "completions/mean_length": 45.375, "completions/min_length": 38.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.375, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.8829458951950073, "rewards/meter/std": 0.08990895003080368, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9137102961540222, "rewards/repeat_soft/std": 0.03713247925043106, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.5805026292800903, "rewards/total_composite/std": 0.030275192111730576, "reward": 0.5805026292800903, "reward_std": 0.03027520515024662, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11579596996307373, "sampling/sampling_logp_difference/max": 1.4301106929779053, "sampling/importance_sampling_ratio/min": 0.23928244411945343, "sampling/importance_sampling_ratio/mean": 0.9996203780174255, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7600813508033752, "clip_ratio/low_mean": 0.04526778403669596, "clip_ratio/low_min": 0.04526778403669596, "clip_ratio/high_mean": 0.05815478228032589, "clip_ratio/high_max": 0.05815478228032589, "clip_ratio/region_mean": 0.10342256631702185, "reward_total_mean": 0.5805026292800903, "reward_meter_mean": 0.8829458951950073, "reward_meter_std": 0.08990895003080368, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9137102961540222, "reward_repeat_soft_std": 0.03713247925043106, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.5805026292800903, "reward_total_composite_std": 0.030275192111730576} {"timestamp_utc": "2026-04-13T11:35:15Z", "mode": "train", "global_step": 1722, "epoch": 0.17297840281265695, "loss": 0.1096, "grad_norm": 15.155242919921875, "learning_rate": 4.784848484848485e-06, "num_tokens": 3064764.0, "completions/mean_length": 32.625, "completions/min_length": 26.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.625, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.6672650575637817, "rewards/meter/std": 0.4079885482788086, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9153906106948853, "rewards/repeat_soft/std": 0.05205296352505684, "rewards/judge_quality/mean": 0.5199999809265137, "rewards/judge_quality/std": 0.19272483885288239, "rewards/total_composite/mean": 0.5362443923950195, "rewards/total_composite/std": 0.10083714127540588, "reward": 0.5362443923950195, "reward_std": 0.10083714127540588, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12003348022699356, "sampling/sampling_logp_difference/max": 2.377509593963623, "sampling/importance_sampling_ratio/min": 0.09278135746717453, "sampling/importance_sampling_ratio/mean": 0.9632998704910278, "sampling/importance_sampling_ratio/max": 1.7206178903579712, "entropy": 0.4664684534072876, "clip_ratio/low_mean": 0.01881533139385283, "clip_ratio/low_min": 0.01881533139385283, "clip_ratio/high_mean": 0.10188171919435263, "clip_ratio/high_max": 0.10188171919435263, "clip_ratio/region_mean": 0.12069705058820546, "reward_total_mean": 0.5362443923950195, "reward_meter_mean": 0.6672650575637817, "reward_meter_std": 0.4079885482788086, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9153906106948853, "reward_repeat_soft_std": 0.05205296352505684, "reward_judge_quality_mean": 0.5199999809265137, "reward_judge_quality_std": 0.19272483885288239, "reward_total_composite_mean": 0.5362443923950195, "reward_total_composite_std": 0.10083714127540588} {"timestamp_utc": "2026-04-13T11:35:22Z", "mode": "train", "global_step": 1723, "epoch": 0.17307885484681065, "loss": 0.0051, "grad_norm": 13.15866756439209, "learning_rate": 4.7818181818181825e-06, "num_tokens": 3066345.0, "completions/mean_length": 43.625, "completions/min_length": 37.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.625, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.7076498866081238, "rewards/meter/std": 0.28890377283096313, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9060574769973755, "rewards/repeat_soft/std": 0.05195293948054314, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.5321946740150452, "rewards/total_composite/std": 0.0805114358663559, "reward": 0.5321946740150452, "reward_std": 0.0805114284157753, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10051216930150986, "sampling/sampling_logp_difference/max": 2.5790698528289795, "sampling/importance_sampling_ratio/min": 0.07584451884031296, "sampling/importance_sampling_ratio/mean": 1.0028880834579468, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5359084531664848, "clip_ratio/low_mean": 0.030447634868323803, "clip_ratio/low_min": 0.030447634868323803, "clip_ratio/high_mean": 0.05426136334426701, "clip_ratio/high_max": 0.05426136334426701, "clip_ratio/region_mean": 0.08470899821259081, "reward_total_mean": 0.5321946740150452, "reward_meter_mean": 0.7076498866081238, "reward_meter_std": 0.28890377283096313, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9060574769973755, "reward_repeat_soft_std": 0.05195293948054314, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.5321946740150452, "reward_total_composite_std": 0.0805114358663559} {"timestamp_utc": "2026-04-13T11:35:29Z", "mode": "train", "global_step": 1724, "epoch": 0.17317930688096433, "loss": 0.0253, "grad_norm": 14.858102798461914, "learning_rate": 4.77878787878788e-06, "num_tokens": 3067837.0, "completions/mean_length": 40.5, "completions/min_length": 36.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.5, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.7685610055923462, "rewards/meter/std": 0.25966876745224, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9639634490013123, "rewards/repeat_soft/std": 0.032831985503435135, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.5945678353309631, "rewards/total_composite/std": 0.15389123558998108, "reward": 0.5945678353309631, "reward_std": 0.15389125049114227, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12579135596752167, "sampling/sampling_logp_difference/max": 1.070995807647705, "sampling/importance_sampling_ratio/min": 0.3426671028137207, "sampling/importance_sampling_ratio/mean": 1.0026395320892334, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7431009709835052, "clip_ratio/low_mean": 0.08824184723198414, "clip_ratio/low_min": 0.08824184723198414, "clip_ratio/high_mean": 0.0321815712377429, "clip_ratio/high_max": 0.0321815712377429, "clip_ratio/region_mean": 0.12042341846972704, "reward_total_mean": 0.5945678353309631, "reward_meter_mean": 0.7685610055923462, "reward_meter_std": 0.25966876745224, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9639634490013123, "reward_repeat_soft_std": 0.032831985503435135, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.5945678353309631, "reward_total_composite_std": 0.15389123558998108} {"timestamp_utc": "2026-04-13T11:35:36Z", "mode": "train", "global_step": 1725, "epoch": 0.17327975891511804, "loss": 0.0081, "grad_norm": 15.199424743652344, "learning_rate": 4.775757575757576e-06, "num_tokens": 3069466.0, "completions/mean_length": 35.625, "completions/min_length": 34.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.625, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.7659406661987305, "rewards/meter/std": 0.12717434763908386, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8684053421020508, "rewards/repeat_soft/std": 0.021665874868631363, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.5410131216049194, "rewards/total_composite/std": 0.03264806792140007, "reward": 0.5410131216049194, "reward_std": 0.03264806047081947, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07706625759601593, "sampling/sampling_logp_difference/max": 1.8642072677612305, "sampling/importance_sampling_ratio/min": 0.15501905977725983, "sampling/importance_sampling_ratio/mean": 0.9940878748893738, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.21618294902145863, "clip_ratio/low_mean": 0.03824948891997337, "clip_ratio/low_min": 0.03824948891997337, "clip_ratio/high_mean": 0.010416666744276881, "clip_ratio/high_max": 0.010416666744276881, "clip_ratio/region_mean": 0.048666155664250255, "reward_total_mean": 0.5410131216049194, "reward_meter_mean": 0.7659406661987305, "reward_meter_std": 0.12717434763908386, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8684053421020508, "reward_repeat_soft_std": 0.021665874868631363, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.5410131216049194, "reward_total_composite_std": 0.03264806792140007} {"timestamp_utc": "2026-04-13T11:35:44Z", "mode": "train", "global_step": 1726, "epoch": 0.17338021094927172, "loss": 0.0364, "grad_norm": 9.00948429107666, "learning_rate": 4.772727272727273e-06, "num_tokens": 3071996.0, "completions/mean_length": 111.25, "completions/min_length": 99.0, "completions/max_length": 126.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 111.25, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 126.0, "rewards/meter/mean": 0.790528416633606, "rewards/meter/std": 0.3705079257488251, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.818536102771759, "rewards/repeat_soft/std": 0.13772985339164734, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5385947227478027, "rewards/total_composite/std": 0.09109405428171158, "reward": 0.5385947227478027, "reward_std": 0.09109406173229218, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08634468913078308, "sampling/sampling_logp_difference/max": 1.504404067993164, "sampling/importance_sampling_ratio/min": 0.22214964032173157, "sampling/importance_sampling_ratio/mean": 1.007115364074707, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.49379777163267136, "clip_ratio/low_mean": 0.01975108217447996, "clip_ratio/low_min": 0.01975108217447996, "clip_ratio/high_mean": 0.06179565005004406, "clip_ratio/high_max": 0.06179565005004406, "clip_ratio/region_mean": 0.08154673222452402, "reward_total_mean": 0.5385947227478027, "reward_meter_mean": 0.790528416633606, "reward_meter_std": 0.3705079257488251, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.818536102771759, "reward_repeat_soft_std": 0.13772985339164734, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5385947227478027, "reward_total_composite_std": 0.09109405428171158} {"timestamp_utc": "2026-04-13T11:35:52Z", "mode": "train", "global_step": 1727, "epoch": 0.1734806629834254, "loss": 0.0281, "grad_norm": 5.4726033210754395, "learning_rate": 4.769696969696971e-06, "num_tokens": 3074126.0, "completions/mean_length": 108.25, "completions/min_length": 86.0, "completions/max_length": 144.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.25, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 144.0, "rewards/meter/mean": 0.9783064723014832, "rewards/meter/std": 0.007912681438028812, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7211712598800659, "rewards/repeat_soft/std": 0.06373512744903564, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.10507651418447495, "rewards/total_composite/mean": 0.6012626886367798, "rewards/total_composite/std": 0.06286019831895828, "reward": 0.6012626886367798, "reward_std": 0.06286019831895828, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08809080719947815, "sampling/sampling_logp_difference/max": 1.8019466400146484, "sampling/importance_sampling_ratio/min": 0.25487732887268066, "sampling/importance_sampling_ratio/mean": 1.01508367061615, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5446457751095295, "clip_ratio/low_mean": 0.06396431918255985, "clip_ratio/low_min": 0.06396431918255985, "clip_ratio/high_mean": 0.012019230984151363, "clip_ratio/high_max": 0.012019230984151363, "clip_ratio/region_mean": 0.07598355016671121, "reward_total_mean": 0.6012626886367798, "reward_meter_mean": 0.9783064723014832, "reward_meter_std": 0.007912681438028812, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7211712598800659, "reward_repeat_soft_std": 0.06373512744903564, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.10507651418447495, "reward_total_composite_mean": 0.6012626886367798, "reward_total_composite_std": 0.06286019831895828} {"timestamp_utc": "2026-04-13T11:35:59Z", "mode": "train", "global_step": 1728, "epoch": 0.1735811150175791, "loss": -0.0284, "grad_norm": 10.919466018676758, "learning_rate": 4.766666666666667e-06, "num_tokens": 3075944.0, "completions/mean_length": 51.25, "completions/min_length": 42.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.25, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.8725104928016663, "rewards/meter/std": 0.17084768414497375, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8668285608291626, "rewards/repeat_soft/std": 0.0349387563765049, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.5713319778442383, "rewards/total_composite/std": 0.043708302080631256, "reward": 0.5713319778442383, "reward_std": 0.043708302080631256, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09125392138957977, "sampling/sampling_logp_difference/max": 1.3393590450286865, "sampling/importance_sampling_ratio/min": 0.2620135545730591, "sampling/importance_sampling_ratio/mean": 1.034496784210205, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6422306261956692, "clip_ratio/low_mean": 0.03758142422884703, "clip_ratio/low_min": 0.03758142422884703, "clip_ratio/high_mean": 0.0488913943991065, "clip_ratio/high_max": 0.0488913943991065, "clip_ratio/region_mean": 0.08647281862795353, "reward_total_mean": 0.5713319778442383, "reward_meter_mean": 0.8725104928016663, "reward_meter_std": 0.17084768414497375, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8668285608291626, "reward_repeat_soft_std": 0.0349387563765049, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.5713319778442383, "reward_total_composite_std": 0.043708302080631256} {"timestamp_utc": "2026-04-13T11:36:06Z", "mode": "train", "global_step": 1729, "epoch": 0.1736815670517328, "loss": -0.0457, "grad_norm": 15.969704627990723, "learning_rate": 4.763636363636364e-06, "num_tokens": 3077334.0, "completions/mean_length": 22.75, "completions/min_length": 16.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.75, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.8908631801605225, "rewards/meter/std": 0.20336231589317322, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9284578561782837, "rewards/repeat_soft/std": 0.034619759768247604, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.5687318444252014, "rewards/total_composite/std": 0.05835019424557686, "reward": 0.5687318444252014, "reward_std": 0.05835019052028656, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08931823819875717, "sampling/sampling_logp_difference/max": 0.7789598703384399, "sampling/importance_sampling_ratio/min": 0.47100359201431274, "sampling/importance_sampling_ratio/mean": 1.0275490283966064, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7520938664674759, "clip_ratio/low_mean": 0.011363636702299118, "clip_ratio/low_min": 0.011363636702299118, "clip_ratio/high_mean": 0.05802327301353216, "clip_ratio/high_max": 0.05802327301353216, "clip_ratio/region_mean": 0.06938690971583128, "reward_total_mean": 0.5687318444252014, "reward_meter_mean": 0.8908631801605225, "reward_meter_std": 0.20336231589317322, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9284578561782837, "reward_repeat_soft_std": 0.034619759768247604, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.5687318444252014, "reward_total_composite_std": 0.05835019424557686} {"timestamp_utc": "2026-04-13T11:36:12Z", "mode": "train", "global_step": 1730, "epoch": 0.1737820190858865, "loss": 0.1095, "grad_norm": 12.536210060119629, "learning_rate": 4.760606060606061e-06, "num_tokens": 3078969.0, "completions/mean_length": 38.375, "completions/min_length": 33.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.375, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.49436667561531067, "rewards/meter/std": 0.4479481875896454, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8966197371482849, "rewards/repeat_soft/std": 0.059394724667072296, "rewards/judge_quality/mean": 0.4387499988079071, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.4764440953731537, "rewards/total_composite/std": 0.13087791204452515, "reward": 0.4764440953731537, "reward_std": 0.13087791204452515, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11847681552171707, "sampling/sampling_logp_difference/max": 1.4087927341461182, "sampling/importance_sampling_ratio/min": 0.24443821609020233, "sampling/importance_sampling_ratio/mean": 1.0156583786010742, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.568436685949564, "clip_ratio/low_mean": 0.04864378087222576, "clip_ratio/low_min": 0.04864378087222576, "clip_ratio/high_mean": 0.048414388904348016, "clip_ratio/high_max": 0.048414388904348016, "clip_ratio/region_mean": 0.09705816977657378, "reward_total_mean": 0.4764440953731537, "reward_meter_mean": 0.49436667561531067, "reward_meter_std": 0.4479481875896454, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8966197371482849, "reward_repeat_soft_std": 0.059394724667072296, "reward_judge_quality_mean": 0.4387499988079071, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.4764440953731537, "reward_total_composite_std": 0.13087791204452515} {"timestamp_utc": "2026-04-13T11:36:20Z", "mode": "train", "global_step": 1731, "epoch": 0.17388247112004018, "loss": 0.0197, "grad_norm": 4.172364711761475, "learning_rate": 4.757575757575758e-06, "num_tokens": 3081969.0, "completions/mean_length": 166.0, "completions/min_length": 120.0, "completions/max_length": 189.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 166.0, "completions/min_terminated_length": 120.0, "completions/max_terminated_length": 189.0, "rewards/meter/mean": 0.9715299606323242, "rewards/meter/std": 0.010508740320801735, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.08625820279121399, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.660871684551239, "rewards/repeat_soft/std": 0.10618186742067337, "rewards/judge_quality/mean": 0.4025000035762787, "rewards/judge_quality/std": 0.06250713765621185, "rewards/total_composite/mean": 0.5408867597579956, "rewards/total_composite/std": 0.04914843291044235, "reward": 0.5408867597579956, "reward_std": 0.049148425459861755, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07430362701416016, "sampling/sampling_logp_difference/max": 9.649591445922852, "sampling/importance_sampling_ratio/min": 6.445188773795962e-05, "sampling/importance_sampling_ratio/mean": 1.0056936740875244, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.34446943178772926, "clip_ratio/low_mean": 0.011553122196346521, "clip_ratio/low_min": 0.011553122196346521, "clip_ratio/high_mean": 0.04789688950404525, "clip_ratio/high_max": 0.04789688950404525, "clip_ratio/region_mean": 0.05945001170039177, "reward_total_mean": 0.5408867597579956, "reward_meter_mean": 0.9715299606323242, "reward_meter_std": 0.010508740320801735, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.08625820279121399, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.660871684551239, "reward_repeat_soft_std": 0.10618186742067337, "reward_judge_quality_mean": 0.4025000035762787, "reward_judge_quality_std": 0.06250713765621185, "reward_total_composite_mean": 0.5408867597579956, "reward_total_composite_std": 0.04914843291044235} {"timestamp_utc": "2026-04-13T11:36:27Z", "mode": "train", "global_step": 1732, "epoch": 0.17398292315419386, "loss": 0.045, "grad_norm": 8.757774353027344, "learning_rate": 4.754545454545455e-06, "num_tokens": 3083735.0, "completions/mean_length": 58.75, "completions/min_length": 54.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.75, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9289580583572388, "rewards/meter/std": 0.052348531782627106, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8491862416267395, "rewards/repeat_soft/std": 0.05479966104030609, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.5675961971282959, "rewards/total_composite/std": 0.04271506518125534, "reward": 0.5675961971282959, "reward_std": 0.04271508380770683, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07935873419046402, "sampling/sampling_logp_difference/max": 1.4487967491149902, "sampling/importance_sampling_ratio/min": 0.23485270142555237, "sampling/importance_sampling_ratio/mean": 1.002606987953186, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.33651405945420265, "clip_ratio/low_mean": 0.023050089366734028, "clip_ratio/low_min": 0.023050089366734028, "clip_ratio/high_mean": 0.045291122049093246, "clip_ratio/high_max": 0.045291122049093246, "clip_ratio/region_mean": 0.06834121141582727, "reward_total_mean": 0.5675961971282959, "reward_meter_mean": 0.9289580583572388, "reward_meter_std": 0.052348531782627106, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8491862416267395, "reward_repeat_soft_std": 0.05479966104030609, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.5675961971282959, "reward_total_composite_std": 0.04271506518125534} {"timestamp_utc": "2026-04-13T11:36:34Z", "mode": "train", "global_step": 1733, "epoch": 0.17408337518834757, "loss": -0.0296, "grad_norm": 8.731679916381836, "learning_rate": 4.751515151515152e-06, "num_tokens": 3085653.0, "completions/mean_length": 61.75, "completions/min_length": 55.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.75, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.7169671058654785, "rewards/meter/std": 0.36305248737335205, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8858823180198669, "rewards/repeat_soft/std": 0.0730840340256691, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.5934597253799438, "rewards/total_composite/std": 0.16547562181949615, "reward": 0.5934597253799438, "reward_std": 0.16547560691833496, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07510776072740555, "sampling/sampling_logp_difference/max": 1.4382802248001099, "sampling/importance_sampling_ratio/min": 0.23733556270599365, "sampling/importance_sampling_ratio/mean": 1.0047154426574707, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.305077251046896, "clip_ratio/low_mean": 0.048390478594228625, "clip_ratio/low_min": 0.048390478594228625, "clip_ratio/high_mean": 0.027283618226647377, "clip_ratio/high_max": 0.027283618226647377, "clip_ratio/region_mean": 0.075674096820876, "reward_total_mean": 0.5934597253799438, "reward_meter_mean": 0.7169671058654785, "reward_meter_std": 0.36305248737335205, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8858823180198669, "reward_repeat_soft_std": 0.0730840340256691, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.5934597253799438, "reward_total_composite_std": 0.16547562181949615} {"timestamp_utc": "2026-04-13T11:36:42Z", "mode": "train", "global_step": 1734, "epoch": 0.17418382722250125, "loss": 0.0396, "grad_norm": 9.175833702087402, "learning_rate": 4.748484848484849e-06, "num_tokens": 3087311.0, "completions/mean_length": 57.25, "completions/min_length": 50.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.25, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9507708549499512, "rewards/meter/std": 0.06551821529865265, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.923712968826294, "rewards/repeat_soft/std": 0.03284953162074089, "rewards/judge_quality/mean": 0.5487500429153442, "rewards/judge_quality/std": 0.22937415540218353, "rewards/total_composite/mean": 0.6728672385215759, "rewards/total_composite/std": 0.1251244693994522, "reward": 0.6728672385215759, "reward_std": 0.1251244693994522, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09673381596803665, "sampling/sampling_logp_difference/max": 0.9068347215652466, "sampling/importance_sampling_ratio/min": 0.4449917674064636, "sampling/importance_sampling_ratio/mean": 1.0263419151306152, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.643095389008522, "clip_ratio/low_mean": 0.08720999862998724, "clip_ratio/low_min": 0.08720999862998724, "clip_ratio/high_mean": 0.02271186374127865, "clip_ratio/high_max": 0.02271186374127865, "clip_ratio/region_mean": 0.10992186237126589, "reward_total_mean": 0.6728672385215759, "reward_meter_mean": 0.9507708549499512, "reward_meter_std": 0.06551821529865265, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.923712968826294, "reward_repeat_soft_std": 0.03284953162074089, "reward_judge_quality_mean": 0.5487500429153442, "reward_judge_quality_std": 0.22937415540218353, "reward_total_composite_mean": 0.6728672385215759, "reward_total_composite_std": 0.1251244693994522} {"timestamp_utc": "2026-04-13T11:36:49Z", "mode": "train", "global_step": 1735, "epoch": 0.17428427925665493, "loss": -0.0233, "grad_norm": 11.077441215515137, "learning_rate": 4.745454545454546e-06, "num_tokens": 3088892.0, "completions/mean_length": 47.625, "completions/min_length": 41.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.625, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9870283007621765, "rewards/meter/std": 0.009919442236423492, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9679203629493713, "rewards/repeat_soft/std": 0.039826054126024246, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.6194697618484497, "rewards/total_composite/std": 0.01266756933182478, "reward": 0.6194697618484497, "reward_std": 0.012667574919760227, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09732978045940399, "sampling/sampling_logp_difference/max": 1.9471259117126465, "sampling/importance_sampling_ratio/min": 0.14268356561660767, "sampling/importance_sampling_ratio/mean": 1.0145297050476074, "sampling/importance_sampling_ratio/max": 1.7899833917617798, "entropy": 0.7135571166872978, "clip_ratio/low_mean": 0.04798766737803817, "clip_ratio/low_min": 0.04798766737803817, "clip_ratio/high_mean": 0.04465630929917097, "clip_ratio/high_max": 0.04465630929917097, "clip_ratio/region_mean": 0.09264397667720914, "reward_total_mean": 0.6194697618484497, "reward_meter_mean": 0.9870283007621765, "reward_meter_std": 0.009919442236423492, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9679203629493713, "reward_repeat_soft_std": 0.039826054126024246, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.6194697618484497, "reward_total_composite_std": 0.01266756933182478} {"timestamp_utc": "2026-04-13T11:36:56Z", "mode": "train", "global_step": 1736, "epoch": 0.17438473129080864, "loss": 0.0929, "grad_norm": 7.1619768142700195, "learning_rate": 4.7424242424242426e-06, "num_tokens": 3091085.0, "completions/mean_length": 88.125, "completions/min_length": 66.0, "completions/max_length": 121.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 88.125, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.9182576537132263, "rewards/meter/std": 0.07804084569215775, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7472658753395081, "rewards/repeat_soft/std": 0.1533329039812088, "rewards/judge_quality/mean": 0.41499999165534973, "rewards/judge_quality/std": 0.14520922303199768, "rewards/total_composite/mean": 0.557198703289032, "rewards/total_composite/std": 0.08052248507738113, "reward": 0.557198703289032, "reward_std": 0.08052248507738113, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08153990656137466, "sampling/sampling_logp_difference/max": 1.4716401100158691, "sampling/importance_sampling_ratio/min": 0.22954869270324707, "sampling/importance_sampling_ratio/mean": 1.0046273469924927, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5766340754926205, "clip_ratio/low_mean": 0.04738907259888947, "clip_ratio/low_min": 0.04738907259888947, "clip_ratio/high_mean": 0.029976489953696728, "clip_ratio/high_max": 0.029976489953696728, "clip_ratio/region_mean": 0.0773655625525862, "reward_total_mean": 0.557198703289032, "reward_meter_mean": 0.9182576537132263, "reward_meter_std": 0.07804084569215775, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7472658753395081, "reward_repeat_soft_std": 0.1533329039812088, "reward_judge_quality_mean": 0.41499999165534973, "reward_judge_quality_std": 0.14520922303199768, "reward_total_composite_mean": 0.557198703289032, "reward_total_composite_std": 0.08052248507738113} {"timestamp_utc": "2026-04-13T11:37:03Z", "mode": "train", "global_step": 1737, "epoch": 0.17448518332496232, "loss": 0.122, "grad_norm": 15.391199111938477, "learning_rate": 4.73939393939394e-06, "num_tokens": 3092727.0, "completions/mean_length": 51.25, "completions/min_length": 38.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.25, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.6474987268447876, "rewards/meter/std": 0.4506898522377014, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9824091196060181, "rewards/repeat_soft/std": 0.030868539586663246, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.5271519422531128, "rewards/total_composite/std": 0.12375441193580627, "reward": 0.5271519422531128, "reward_std": 0.12375440448522568, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14935629069805145, "sampling/sampling_logp_difference/max": 1.8774235248565674, "sampling/importance_sampling_ratio/min": 0.15298376977443695, "sampling/importance_sampling_ratio/mean": 1.0210734605789185, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.114188238978386, "clip_ratio/low_mean": 0.044731407426297665, "clip_ratio/low_min": 0.044731407426297665, "clip_ratio/high_mean": 0.07409316394478083, "clip_ratio/high_max": 0.07409316394478083, "clip_ratio/region_mean": 0.11882457137107849, "reward_total_mean": 0.5271519422531128, "reward_meter_mean": 0.6474987268447876, "reward_meter_std": 0.4506898522377014, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9824091196060181, "reward_repeat_soft_std": 0.030868539586663246, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.5271519422531128, "reward_total_composite_std": 0.12375441193580627} {"timestamp_utc": "2026-04-13T11:37:11Z", "mode": "train", "global_step": 1738, "epoch": 0.17458563535911603, "loss": 0.0749, "grad_norm": 9.294628143310547, "learning_rate": 4.736363636363637e-06, "num_tokens": 3095324.0, "completions/mean_length": 113.625, "completions/min_length": 102.0, "completions/max_length": 130.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 113.625, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 130.0, "rewards/meter/mean": 0.9556190967559814, "rewards/meter/std": 0.10261744260787964, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7098491191864014, "rewards/repeat_soft/std": 0.07084794342517853, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.0645727664232254, "rewards/total_composite/mean": 0.5585969686508179, "rewards/total_composite/std": 0.05751483142375946, "reward": 0.5585969686508179, "reward_std": 0.05751482769846916, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08325659483671188, "sampling/sampling_logp_difference/max": 1.6753778457641602, "sampling/importance_sampling_ratio/min": 0.18723741173744202, "sampling/importance_sampling_ratio/mean": 0.9986563920974731, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3381666596978903, "clip_ratio/low_mean": 0.01459030108526349, "clip_ratio/low_min": 0.01459030108526349, "clip_ratio/high_mean": 0.06721780961379409, "clip_ratio/high_max": 0.06721780961379409, "clip_ratio/region_mean": 0.08180811069905758, "reward_total_mean": 0.5585969686508179, "reward_meter_mean": 0.9556190967559814, "reward_meter_std": 0.10261744260787964, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7098491191864014, "reward_repeat_soft_std": 0.07084794342517853, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.0645727664232254, "reward_total_composite_mean": 0.5585969686508179, "reward_total_composite_std": 0.05751483142375946} {"timestamp_utc": "2026-04-13T11:37:18Z", "mode": "train", "global_step": 1739, "epoch": 0.1746860873932697, "loss": 0.033, "grad_norm": 15.374832153320312, "learning_rate": 4.7333333333333335e-06, "num_tokens": 3096913.0, "completions/mean_length": 33.625, "completions/min_length": 30.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.625, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.8024061918258667, "rewards/meter/std": 0.32132992148399353, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8930982947349548, "rewards/repeat_soft/std": 0.02109317108988762, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.267261266708374, "rewards/total_composite/mean": 0.6610607504844666, "rewards/total_composite/std": 0.18794965744018555, "reward": 0.6610607504844666, "reward_std": 0.18794964253902435, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.076873779296875, "sampling/sampling_logp_difference/max": 1.4227275848388672, "sampling/importance_sampling_ratio/min": 0.24105563759803772, "sampling/importance_sampling_ratio/mean": 1.011893391609192, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4238227345049381, "clip_ratio/low_mean": 0.05576387792825699, "clip_ratio/low_min": 0.05576387792825699, "clip_ratio/high_mean": 0.022368420846760273, "clip_ratio/high_max": 0.022368420846760273, "clip_ratio/region_mean": 0.07813229877501726, "reward_total_mean": 0.6610607504844666, "reward_meter_mean": 0.8024061918258667, "reward_meter_std": 0.32132992148399353, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8930982947349548, "reward_repeat_soft_std": 0.02109317108988762, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.267261266708374, "reward_total_composite_mean": 0.6610607504844666, "reward_total_composite_std": 0.18794965744018555} {"timestamp_utc": "2026-04-13T11:37:26Z", "mode": "train", "global_step": 1740, "epoch": 0.1747865394274234, "loss": -0.0102, "grad_norm": 5.100631237030029, "learning_rate": 4.730303030303031e-06, "num_tokens": 3099420.0, "completions/mean_length": 126.375, "completions/min_length": 107.0, "completions/max_length": 147.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 126.375, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 147.0, "rewards/meter/mean": 0.9850823879241943, "rewards/meter/std": 0.007084795273840427, "rewards/count_adherence/mean": 0.8500000238418579, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7304711937904358, "rewards/repeat_soft/std": 0.09397655725479126, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.10925068706274033, "rewards/total_composite/mean": 0.4860914349555969, "rewards/total_composite/std": 0.07439238578081131, "reward": 0.4860914349555969, "reward_std": 0.07439238578081131, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.061496373265981674, "sampling/sampling_logp_difference/max": 1.6928315162658691, "sampling/importance_sampling_ratio/min": 0.18399780988693237, "sampling/importance_sampling_ratio/mean": 1.0103132724761963, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3876885324716568, "clip_ratio/low_mean": 0.022239373065531254, "clip_ratio/low_min": 0.022239373065531254, "clip_ratio/high_mean": 0.028220489853993058, "clip_ratio/high_max": 0.028220489853993058, "clip_ratio/region_mean": 0.05045986291952431, "reward_total_mean": 0.4860914349555969, "reward_meter_mean": 0.9850823879241943, "reward_meter_std": 0.007084795273840427, "reward_count_adherence_mean": 0.8500000238418579, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7304711937904358, "reward_repeat_soft_std": 0.09397655725479126, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.10925068706274033, "reward_total_composite_mean": 0.4860914349555969, "reward_total_composite_std": 0.07439238578081131} {"timestamp_utc": "2026-04-13T11:37:33Z", "mode": "train", "global_step": 1741, "epoch": 0.1748869914615771, "loss": -0.0792, "grad_norm": 12.408956527709961, "learning_rate": 4.727272727272728e-06, "num_tokens": 3101128.0, "completions/mean_length": 50.5, "completions/min_length": 42.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.5, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.4509848952293396, "rewards/meter/std": 0.30584847927093506, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.991193413734436, "rewards/repeat_soft/std": 0.009842580184340477, "rewards/judge_quality/mean": 0.59375, "rewards/judge_quality/std": 0.2775370180606842, "rewards/total_composite/mean": 0.5134477019309998, "rewards/total_composite/std": 0.13883522152900696, "reward": 0.5134477019309998, "reward_std": 0.13883522152900696, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10810422897338867, "sampling/sampling_logp_difference/max": 1.3345818519592285, "sampling/importance_sampling_ratio/min": 0.26326823234558105, "sampling/importance_sampling_ratio/mean": 1.00209641456604, "sampling/importance_sampling_ratio/max": 1.9134043455123901, "entropy": 0.44106733053922653, "clip_ratio/low_mean": 0.05428210739046335, "clip_ratio/low_min": 0.05428210739046335, "clip_ratio/high_mean": 0.0437433160841465, "clip_ratio/high_max": 0.0437433160841465, "clip_ratio/region_mean": 0.09802542347460985, "reward_total_mean": 0.5134477019309998, "reward_meter_mean": 0.4509848952293396, "reward_meter_std": 0.30584847927093506, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.991193413734436, "reward_repeat_soft_std": 0.009842580184340477, "reward_judge_quality_mean": 0.59375, "reward_judge_quality_std": 0.2775370180606842, "reward_total_composite_mean": 0.5134477019309998, "reward_total_composite_std": 0.13883522152900696} {"timestamp_utc": "2026-04-13T11:37:42Z", "mode": "train", "global_step": 1742, "epoch": 0.17498744349573078, "loss": 0.0969, "grad_norm": 5.192895889282227, "learning_rate": 4.724242424242424e-06, "num_tokens": 3103687.0, "completions/mean_length": 131.875, "completions/min_length": 98.0, "completions/max_length": 187.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.875, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 187.0, "rewards/meter/mean": 0.8247295022010803, "rewards/meter/std": 0.2682240605354309, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6811105012893677, "rewards/repeat_soft/std": 0.08826851844787598, "rewards/judge_quality/mean": 0.3037499785423279, "rewards/judge_quality/std": 0.13362392783164978, "rewards/total_composite/mean": 0.4580414593219757, "rewards/total_composite/std": 0.09204375743865967, "reward": 0.4580414593219757, "reward_std": 0.09204376488924026, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06757155805826187, "sampling/sampling_logp_difference/max": 1.2468342781066895, "sampling/importance_sampling_ratio/min": 0.28741323947906494, "sampling/importance_sampling_ratio/mean": 1.0102301836013794, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5221900083124638, "clip_ratio/low_mean": 0.022508766502141953, "clip_ratio/low_min": 0.022508766502141953, "clip_ratio/high_mean": 0.027129985624924302, "clip_ratio/high_max": 0.027129985624924302, "clip_ratio/region_mean": 0.049638752127066255, "reward_total_mean": 0.4580414593219757, "reward_meter_mean": 0.8247295022010803, "reward_meter_std": 0.2682240605354309, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6811105012893677, "reward_repeat_soft_std": 0.08826851844787598, "reward_judge_quality_mean": 0.3037499785423279, "reward_judge_quality_std": 0.13362392783164978, "reward_total_composite_mean": 0.4580414593219757, "reward_total_composite_std": 0.09204375743865967} {"timestamp_utc": "2026-04-13T11:37:54Z", "mode": "train", "global_step": 1743, "epoch": 0.1750878955298845, "loss": -0.1299, "grad_norm": 1.7194968461990356, "learning_rate": 4.721212121212122e-06, "num_tokens": 3105323.0, "completions/mean_length": 169.5, "completions/min_length": 47.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 55.333335876464844, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.8567893505096436, "rewards/meter/std": 0.3435137867927551, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.934806764125824, "rewards/repeat_soft/std": 0.044237542897462845, "rewards/judge_quality/mean": 0.32749998569488525, "rewards/judge_quality/std": 0.17127670347690582, "rewards/total_composite/mean": 0.45591509342193604, "rewards/total_composite/std": 0.28141120076179504, "reward": 0.45591509342193604, "reward_std": 0.28141117095947266, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.134735569357872, "sampling/sampling_logp_difference/max": 1.4726686477661133, "sampling/importance_sampling_ratio/min": 0.2293127328157425, "sampling/importance_sampling_ratio/mean": 1.016987919807434, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5698898136615753, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09488785662688315, "clip_ratio/high_max": 0.09488785662688315, "clip_ratio/region_mean": 0.09488785662688315, "reward_total_mean": 0.45591509342193604, "reward_meter_mean": 0.8567893505096436, "reward_meter_std": 0.3435137867927551, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.934806764125824, "reward_repeat_soft_std": 0.044237542897462845, "reward_judge_quality_mean": 0.32749998569488525, "reward_judge_quality_std": 0.17127670347690582, "reward_total_composite_mean": 0.45591509342193604, "reward_total_composite_std": 0.28141120076179504} {"timestamp_utc": "2026-04-13T11:38:00Z", "mode": "train", "global_step": 1744, "epoch": 0.17518834756403817, "loss": 0.0997, "grad_norm": 18.683486938476562, "learning_rate": 4.718181818181818e-06, "num_tokens": 3106680.0, "completions/mean_length": 25.625, "completions/min_length": 18.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.625, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.9062577486038208, "rewards/meter/std": 0.11527059972286224, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9534684419631958, "rewards/repeat_soft/std": 0.021279437467455864, "rewards/judge_quality/mean": 0.44624999165534973, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.6056767702102661, "rewards/total_composite/std": 0.03251388669013977, "reward": 0.6056767702102661, "reward_std": 0.032513879239559174, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14203035831451416, "sampling/sampling_logp_difference/max": 1.0328145027160645, "sampling/importance_sampling_ratio/min": 0.3560035824775696, "sampling/importance_sampling_ratio/mean": 1.0251483917236328, "sampling/importance_sampling_ratio/max": 1.7394042015075684, "entropy": 1.2420547008514404, "clip_ratio/low_mean": 0.023888888768851757, "clip_ratio/low_min": 0.023888888768851757, "clip_ratio/high_mean": 0.09716971544548869, "clip_ratio/high_max": 0.09716971544548869, "clip_ratio/region_mean": 0.12105860421434045, "reward_total_mean": 0.6056767702102661, "reward_meter_mean": 0.9062577486038208, "reward_meter_std": 0.11527059972286224, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9534684419631958, "reward_repeat_soft_std": 0.021279437467455864, "reward_judge_quality_mean": 0.44624999165534973, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.6056767702102661, "reward_total_composite_std": 0.03251388669013977} {"timestamp_utc": "2026-04-13T11:38:06Z", "mode": "train", "global_step": 1745, "epoch": 0.17528879959819185, "loss": 0.0503, "grad_norm": 14.6497802734375, "learning_rate": 4.715151515151515e-06, "num_tokens": 3108114.0, "completions/mean_length": 24.25, "completions/min_length": 19.0, "completions/max_length": 27.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.25, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.9924908876419067, "rewards/meter/std": 0.003887118538841605, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9176196455955505, "rewards/repeat_soft/std": 0.052994225174188614, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.6182574033737183, "rewards/total_composite/std": 0.012917676940560341, "reward": 0.6182574033737183, "reward_std": 0.012917687185108662, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10061626881361008, "sampling/sampling_logp_difference/max": 1.249585509300232, "sampling/importance_sampling_ratio/min": 0.28662359714508057, "sampling/importance_sampling_ratio/mean": 1.0225293636322021, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9781959131360054, "clip_ratio/low_mean": 0.06482456158846617, "clip_ratio/low_min": 0.06482456158846617, "clip_ratio/high_mean": 0.030974460765719414, "clip_ratio/high_max": 0.030974460765719414, "clip_ratio/region_mean": 0.09579902235418558, "reward_total_mean": 0.6182574033737183, "reward_meter_mean": 0.9924908876419067, "reward_meter_std": 0.003887118538841605, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9176196455955505, "reward_repeat_soft_std": 0.052994225174188614, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.6182574033737183, "reward_total_composite_std": 0.012917676940560341} {"timestamp_utc": "2026-04-13T11:38:13Z", "mode": "train", "global_step": 1746, "epoch": 0.17538925163234556, "loss": 0.0244, "grad_norm": 6.043193340301514, "learning_rate": 4.7121212121212126e-06, "num_tokens": 3110838.0, "completions/mean_length": 126.5, "completions/min_length": 107.0, "completions/max_length": 148.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 126.5, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 148.0, "rewards/meter/mean": 0.9032549262046814, "rewards/meter/std": 0.1314523071050644, "rewards/count_adherence/mean": 0.8250000476837158, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7391740083694458, "rewards/repeat_soft/std": 0.1549014300107956, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.1249857097864151, "rewards/total_composite/mean": 0.47975093126296997, "rewards/total_composite/std": 0.0910043939948082, "reward": 0.47975093126296997, "reward_std": 0.0910043865442276, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09540028870105743, "sampling/sampling_logp_difference/max": 1.770749568939209, "sampling/importance_sampling_ratio/min": 0.17020536959171295, "sampling/importance_sampling_ratio/mean": 1.003059983253479, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5396188572049141, "clip_ratio/low_mean": 0.0195392930181697, "clip_ratio/low_min": 0.0195392930181697, "clip_ratio/high_mean": 0.05596997123211622, "clip_ratio/high_max": 0.05596997123211622, "clip_ratio/region_mean": 0.07550926425028592, "reward_total_mean": 0.47975093126296997, "reward_meter_mean": 0.9032549262046814, "reward_meter_std": 0.1314523071050644, "reward_count_adherence_mean": 0.8250000476837158, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7391740083694458, "reward_repeat_soft_std": 0.1549014300107956, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.1249857097864151, "reward_total_composite_mean": 0.47975093126296997, "reward_total_composite_std": 0.0910043939948082} {"timestamp_utc": "2026-04-13T11:38:25Z", "mode": "train", "global_step": 1747, "epoch": 0.17548970366649924, "loss": -0.1828, "grad_norm": 2.840977191925049, "learning_rate": 4.709090909090909e-06, "num_tokens": 3113230.0, "completions/mean_length": 172.0, "completions/min_length": 103.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 123.42857360839844, "completions/min_terminated_length": 103.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.9895341396331787, "rewards/meter/std": 0.007784141227602959, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.971744179725647, "rewards/repeat_soft/std": 0.01658444106578827, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.2308950126171112, "rewards/total_composite/mean": 0.5592749118804932, "rewards/total_composite/std": 0.2532850205898285, "reward": 0.5592749118804932, "reward_std": 0.2532850205898285, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12874051928520203, "sampling/sampling_logp_difference/max": 2.107656478881836, "sampling/importance_sampling_ratio/min": 0.12152242660522461, "sampling/importance_sampling_ratio/mean": 1.0061155557632446, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6671543717384338, "clip_ratio/low_mean": 0.018292502965778112, "clip_ratio/low_min": 0.018292502965778112, "clip_ratio/high_mean": 0.0698564164340496, "clip_ratio/high_max": 0.0698564164340496, "clip_ratio/region_mean": 0.08814891939982772, "reward_total_mean": 0.5592749118804932, "reward_meter_mean": 0.9895341396331787, "reward_meter_std": 0.007784141227602959, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.971744179725647, "reward_repeat_soft_std": 0.01658444106578827, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.2308950126171112, "reward_total_composite_mean": 0.5592749118804932, "reward_total_composite_std": 0.2532850205898285} {"timestamp_utc": "2026-04-13T11:38:32Z", "mode": "train", "global_step": 1748, "epoch": 0.17559015570065295, "loss": 0.0005, "grad_norm": 11.052653312683105, "learning_rate": 4.706060606060606e-06, "num_tokens": 3114991.0, "completions/mean_length": 53.125, "completions/min_length": 50.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.125, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.5705060958862305, "rewards/meter/std": 0.3067757189273834, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8631938099861145, "rewards/repeat_soft/std": 0.0716160461306572, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.4884698987007141, "rewards/total_composite/std": 0.08246802538633347, "reward": 0.4884698987007141, "reward_std": 0.08246801048517227, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1203719899058342, "sampling/sampling_logp_difference/max": 1.6194086074829102, "sampling/importance_sampling_ratio/min": 0.19801577925682068, "sampling/importance_sampling_ratio/mean": 1.0158250331878662, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7180077284574509, "clip_ratio/low_mean": 0.0695974649861455, "clip_ratio/low_min": 0.0695974649861455, "clip_ratio/high_mean": 0.03645964711904526, "clip_ratio/high_max": 0.03645964711904526, "clip_ratio/region_mean": 0.10605711210519075, "reward_total_mean": 0.4884698987007141, "reward_meter_mean": 0.5705060958862305, "reward_meter_std": 0.3067757189273834, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8631938099861145, "reward_repeat_soft_std": 0.0716160461306572, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.4884698987007141, "reward_total_composite_std": 0.08246802538633347} {"timestamp_utc": "2026-04-13T11:38:40Z", "mode": "train", "global_step": 1749, "epoch": 0.17569060773480663, "loss": 0.0629, "grad_norm": 13.392763137817383, "learning_rate": 4.7030303030303035e-06, "num_tokens": 3116363.0, "completions/mean_length": 24.5, "completions/min_length": 21.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.5, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.9797687530517578, "rewards/meter/std": 0.02011459320783615, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9197984933853149, "rewards/repeat_soft/std": 0.05981236323714256, "rewards/judge_quality/mean": 0.39249998331069946, "rewards/judge_quality/std": 0.08892211318016052, "rewards/total_composite/mean": 0.5875974297523499, "rewards/total_composite/std": 0.05380617454648018, "reward": 0.5875974297523499, "reward_std": 0.053806155920028687, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13312344253063202, "sampling/sampling_logp_difference/max": 1.723677635192871, "sampling/importance_sampling_ratio/min": 0.17840883135795593, "sampling/importance_sampling_ratio/mean": 1.022372841835022, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9063016846776009, "clip_ratio/low_mean": 0.06807692348957062, "clip_ratio/low_min": 0.06807692348957062, "clip_ratio/high_mean": 0.09576756693422794, "clip_ratio/high_max": 0.09576756693422794, "clip_ratio/region_mean": 0.16384449042379856, "reward_total_mean": 0.5875974297523499, "reward_meter_mean": 0.9797687530517578, "reward_meter_std": 0.02011459320783615, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9197984933853149, "reward_repeat_soft_std": 0.05981236323714256, "reward_judge_quality_mean": 0.39249998331069946, "reward_judge_quality_std": 0.08892211318016052, "reward_total_composite_mean": 0.5875974297523499, "reward_total_composite_std": 0.05380617454648018} {"timestamp_utc": "2026-04-13T11:38:47Z", "mode": "train", "global_step": 1750, "epoch": 0.1757910597689603, "loss": 0.0417, "grad_norm": 7.338882923126221, "learning_rate": 4.7e-06, "num_tokens": 3118599.0, "completions/mean_length": 98.5, "completions/min_length": 88.0, "completions/max_length": 106.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.5, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.6081505417823792, "rewards/meter/std": 0.31105008721351624, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8411466479301453, "rewards/repeat_soft/std": 0.053297799080610275, "rewards/judge_quality/mean": 0.5575000047683716, "rewards/judge_quality/std": 0.19955308735370636, "rewards/total_composite/mean": 0.5617530941963196, "rewards/total_composite/std": 0.1653565913438797, "reward": 0.5617530941963196, "reward_std": 0.1653565764427185, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09892749786376953, "sampling/sampling_logp_difference/max": 1.4269309043884277, "sampling/importance_sampling_ratio/min": 0.2400445193052292, "sampling/importance_sampling_ratio/mean": 1.0135704278945923, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6103395111858845, "clip_ratio/low_mean": 0.05503624491393566, "clip_ratio/low_min": 0.05503624491393566, "clip_ratio/high_mean": 0.04551269207149744, "clip_ratio/high_max": 0.04551269207149744, "clip_ratio/region_mean": 0.1005489369854331, "reward_total_mean": 0.5617530941963196, "reward_meter_mean": 0.6081505417823792, "reward_meter_std": 0.31105008721351624, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8411466479301453, "reward_repeat_soft_std": 0.053297799080610275, "reward_judge_quality_mean": 0.5575000047683716, "reward_judge_quality_std": 0.19955308735370636, "reward_total_composite_mean": 0.5617530941963196, "reward_total_composite_std": 0.1653565913438797} {"timestamp_utc": "2026-04-13T11:39:47Z", "mode": "eval", "global_step": 1750, "epoch": 0.1757910597689603, "eval_loss": NaN, "eval_runtime": 59.7674, "eval_samples_per_second": 1.339, "eval_steps_per_second": 0.167, "eval_num_tokens": 3118599.0, "eval_completions/mean_length": 93.05, "eval_completions/min_length": 36.2, "eval_completions/max_length": 221.7, "eval_completions/clipped_ratio": 0.025, "eval_completions/mean_terminated_length": 81.93571472167969, "eval_completions/min_terminated_length": 36.2, "eval_completions/max_terminated_length": 143.2, "eval_rewards/meter/mean": 0.7724583685398102, "eval_rewards/meter/std": 0.28240539878606796, "eval_rewards/count_adherence/mean": 0.9633333265781403, "eval_rewards/count_adherence/std": 0.07533912770450116, "eval_rewards/hard_gate/mean": 0.9625, "eval_rewards/hard_gate/std": 0.10606601536273956, "eval_rewards/repeat_soft/mean": 0.862103009223938, "eval_rewards/repeat_soft/std": 0.09512052573263645, "eval_rewards/judge_quality/mean": 0.4583750069141388, "eval_rewards/judge_quality/std": 0.1511122086085379, "eval_rewards/total_composite/mean": 0.5376160174608231, "eval_rewards/total_composite/std": 0.15217913947999478, "eval_reward": 0.5376160174608231, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.049692163988947866, "eval_sampling/sampling_logp_difference/max": 0.954959774017334, "eval_sampling/importance_sampling_ratio/min": 0.3967170387506485, "eval_sampling/importance_sampling_ratio/mean": 1.0106980443000793, "eval_sampling/importance_sampling_ratio/max": 1.4370436906814574, "eval_entropy": 0.5266555488109589, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5376160174608231, "eval_reward_meter_mean": 0.7724583685398102, "eval_reward_meter_std": 0.28240539878606796, "eval_reward_count_adherence_mean": 0.9633333265781403, "eval_reward_count_adherence_std": 0.07533912770450116, "eval_reward_hard_gate_mean": 0.9625, "eval_reward_hard_gate_std": 0.10606601536273956, "eval_reward_repeat_soft_mean": 0.862103009223938, "eval_reward_repeat_soft_std": 0.09512052573263645, "eval_reward_judge_quality_mean": 0.4583750069141388, "eval_reward_judge_quality_std": 0.1511122086085379, "eval_reward_total_composite_mean": 0.5376160174608231, "eval_reward_total_composite_std": 0.15217913947999478} {"timestamp_utc": "2026-04-13T11:39:59Z", "mode": "train", "global_step": 1751, "epoch": 0.17589151180311402, "loss": 0.095, "grad_norm": 6.663627624511719, "learning_rate": 4.696969696969698e-06, "num_tokens": 3120994.0, "completions/mean_length": 123.375, "completions/min_length": 104.0, "completions/max_length": 169.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 123.375, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 169.0, "rewards/meter/mean": 0.9890388250350952, "rewards/meter/std": 0.008543361909687519, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.10690449178218842, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6901615858078003, "rewards/repeat_soft/std": 0.04421984776854515, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.5125491619110107, "rewards/total_composite/std": 0.055993881076574326, "reward": 0.5125491619110107, "reward_std": 0.055993881076574326, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0809365063905716, "sampling/sampling_logp_difference/max": 4.661493301391602, "sampling/importance_sampling_ratio/min": 0.009452336467802525, "sampling/importance_sampling_ratio/mean": 1.0108307600021362, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.41733942180871964, "clip_ratio/low_mean": 0.019566303584724665, "clip_ratio/low_min": 0.019566303584724665, "clip_ratio/high_mean": 0.047922331374138594, "clip_ratio/high_max": 0.047922331374138594, "clip_ratio/region_mean": 0.06748863495886326, "reward_total_mean": 0.5125491619110107, "reward_meter_mean": 0.9890388250350952, "reward_meter_std": 0.008543361909687519, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.10690449178218842, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6901615858078003, "reward_repeat_soft_std": 0.04421984776854515, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.5125491619110107, "reward_total_composite_std": 0.055993881076574326} {"timestamp_utc": "2026-04-13T11:40:05Z", "mode": "train", "global_step": 1752, "epoch": 0.1759919638372677, "loss": -0.0441, "grad_norm": 14.835586547851562, "learning_rate": 4.693939393939394e-06, "num_tokens": 3122487.0, "completions/mean_length": 39.625, "completions/min_length": 33.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.625, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.7192169427871704, "rewards/meter/std": 0.40678057074546814, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9895240068435669, "rewards/repeat_soft/std": 0.014799713157117367, "rewards/judge_quality/mean": 0.48625001311302185, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.5864670276641846, "rewards/total_composite/std": 0.17704661190509796, "reward": 0.5864670276641846, "reward_std": 0.17704661190509796, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11205822974443436, "sampling/sampling_logp_difference/max": 1.7428555488586426, "sampling/importance_sampling_ratio/min": 0.17501990497112274, "sampling/importance_sampling_ratio/mean": 1.0069360733032227, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5704237520694733, "clip_ratio/low_mean": 0.036381255835294724, "clip_ratio/low_min": 0.036381255835294724, "clip_ratio/high_mean": 0.08526288438588381, "clip_ratio/high_max": 0.08526288438588381, "clip_ratio/region_mean": 0.12164414022117853, "reward_total_mean": 0.5864670276641846, "reward_meter_mean": 0.7192169427871704, "reward_meter_std": 0.40678057074546814, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9895240068435669, "reward_repeat_soft_std": 0.014799713157117367, "reward_judge_quality_mean": 0.48625001311302185, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.5864670276641846, "reward_total_composite_std": 0.17704661190509796} {"timestamp_utc": "2026-04-13T11:40:11Z", "mode": "train", "global_step": 1753, "epoch": 0.1760924158714214, "loss": 0.0608, "grad_norm": 15.959066390991211, "learning_rate": 4.690909090909092e-06, "num_tokens": 3123841.0, "completions/mean_length": 21.25, "completions/min_length": 19.0, "completions/max_length": 24.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.25, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 24.0, "rewards/meter/mean": 0.8230654001235962, "rewards/meter/std": 0.3459331691265106, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9470235109329224, "rewards/repeat_soft/std": 0.03704577311873436, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.0645727664232254, "rewards/total_composite/mean": 0.5581273436546326, "rewards/total_composite/std": 0.0982942134141922, "reward": 0.5581273436546326, "reward_std": 0.098294198513031, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14953167736530304, "sampling/sampling_logp_difference/max": 1.1728765964508057, "sampling/importance_sampling_ratio/min": 0.3094754219055176, "sampling/importance_sampling_ratio/mean": 0.9900863170623779, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7450743019580841, "clip_ratio/low_mean": 0.02890037652105093, "clip_ratio/low_min": 0.02890037652105093, "clip_ratio/high_mean": 0.06461187684908509, "clip_ratio/high_max": 0.06461187684908509, "clip_ratio/region_mean": 0.09351225337013602, "reward_total_mean": 0.5581273436546326, "reward_meter_mean": 0.8230654001235962, "reward_meter_std": 0.3459331691265106, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9470235109329224, "reward_repeat_soft_std": 0.03704577311873436, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.0645727664232254, "reward_total_composite_mean": 0.5581273436546326, "reward_total_composite_std": 0.0982942134141922} {"timestamp_utc": "2026-04-13T11:40:23Z", "mode": "train", "global_step": 1754, "epoch": 0.1761928679055751, "loss": -0.1499, "grad_norm": 2.3400802612304688, "learning_rate": 4.687878787878788e-06, "num_tokens": 3125545.0, "completions/mean_length": 123.0, "completions/min_length": 48.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 67.42857360839844, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 97.0, "rewards/meter/mean": 0.32548457384109497, "rewards/meter/std": 0.15839283168315887, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.15430334210395813, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9319171905517578, "rewards/repeat_soft/std": 0.0698607787489891, "rewards/judge_quality/mean": 0.45625001192092896, "rewards/judge_quality/std": 0.20982562005519867, "rewards/total_composite/mean": 0.333653062582016, "rewards/total_composite/std": 0.14164356887340546, "reward": 0.333653062582016, "reward_std": 0.14164355397224426, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14890842139720917, "sampling/sampling_logp_difference/max": 1.6867527961730957, "sampling/importance_sampling_ratio/min": 0.1851196587085724, "sampling/importance_sampling_ratio/mean": 1.0020707845687866, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8648200929164886, "clip_ratio/low_mean": 0.019736841320991516, "clip_ratio/low_min": 0.019736841320991516, "clip_ratio/high_mean": 0.102735442109406, "clip_ratio/high_max": 0.102735442109406, "clip_ratio/region_mean": 0.12247228343039751, "reward_total_mean": 0.333653062582016, "reward_meter_mean": 0.32548457384109497, "reward_meter_std": 0.15839283168315887, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.15430334210395813, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9319171905517578, "reward_repeat_soft_std": 0.0698607787489891, "reward_judge_quality_mean": 0.45625001192092896, "reward_judge_quality_std": 0.20982562005519867, "reward_total_composite_mean": 0.333653062582016, "reward_total_composite_std": 0.14164356887340546} {"timestamp_utc": "2026-04-13T11:40:35Z", "mode": "train", "global_step": 1755, "epoch": 0.17629331993972877, "loss": -0.18, "grad_norm": 3.3729248046875, "learning_rate": 4.684848484848485e-06, "num_tokens": 3127940.0, "completions/mean_length": 172.375, "completions/min_length": 102.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 123.85714721679688, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 146.0, "rewards/meter/mean": 0.8067696690559387, "rewards/meter/std": 0.24337105453014374, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.7807559370994568, "rewards/repeat_soft/std": 0.09562687575817108, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.23445606231689453, "rewards/total_composite/mean": 0.5023723840713501, "rewards/total_composite/std": 0.25182026624679565, "reward": 0.5023723840713501, "reward_std": 0.25182023644447327, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08725365996360779, "sampling/sampling_logp_difference/max": 1.9357357025146484, "sampling/importance_sampling_ratio/min": 0.14431805908679962, "sampling/importance_sampling_ratio/mean": 1.0076463222503662, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.48959536850452423, "clip_ratio/low_mean": 0.02950372127816081, "clip_ratio/low_min": 0.02950372127816081, "clip_ratio/high_mean": 0.03864203952252865, "clip_ratio/high_max": 0.03864203952252865, "clip_ratio/region_mean": 0.06814576080068946, "reward_total_mean": 0.5023723840713501, "reward_meter_mean": 0.8067696690559387, "reward_meter_std": 0.24337105453014374, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.7807559370994568, "reward_repeat_soft_std": 0.09562687575817108, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.23445606231689453, "reward_total_composite_mean": 0.5023723840713501, "reward_total_composite_std": 0.25182026624679565} {"timestamp_utc": "2026-04-13T11:40:42Z", "mode": "train", "global_step": 1756, "epoch": 0.17639377197388248, "loss": -0.0363, "grad_norm": 9.370224952697754, "learning_rate": 4.681818181818183e-06, "num_tokens": 3129970.0, "completions/mean_length": 86.75, "completions/min_length": 71.0, "completions/max_length": 111.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.75, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.36734187602996826, "rewards/meter/std": 0.3477254807949066, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8463608622550964, "rewards/repeat_soft/std": 0.059881750494241714, "rewards/judge_quality/mean": 0.6150000095367432, "rewards/judge_quality/std": 0.21954500675201416, "rewards/total_composite/mean": 0.4815005958080292, "rewards/total_composite/std": 0.17383673787117004, "reward": 0.4815005958080292, "reward_std": 0.17383673787117004, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08453039824962616, "sampling/sampling_logp_difference/max": 2.1572823524475098, "sampling/importance_sampling_ratio/min": 0.11563895642757416, "sampling/importance_sampling_ratio/mean": 1.0083303451538086, "sampling/importance_sampling_ratio/max": 1.9933003187179565, "entropy": 0.41204744577407837, "clip_ratio/low_mean": 0.05621515866369009, "clip_ratio/low_min": 0.05621515866369009, "clip_ratio/high_mean": 0.03240876737982035, "clip_ratio/high_max": 0.03240876737982035, "clip_ratio/region_mean": 0.08862392604351044, "reward_total_mean": 0.4815005958080292, "reward_meter_mean": 0.36734187602996826, "reward_meter_std": 0.3477254807949066, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8463608622550964, "reward_repeat_soft_std": 0.059881750494241714, "reward_judge_quality_mean": 0.6150000095367432, "reward_judge_quality_std": 0.21954500675201416, "reward_total_composite_mean": 0.4815005958080292, "reward_total_composite_std": 0.17383673787117004} {"timestamp_utc": "2026-04-13T11:40:47Z", "mode": "train", "global_step": 1757, "epoch": 0.17649422400803616, "loss": 0.0843, "grad_norm": 16.338285446166992, "learning_rate": 4.678787878787879e-06, "num_tokens": 3131366.0, "completions/mean_length": 23.5, "completions/min_length": 16.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.5, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.8964908719062805, "rewards/meter/std": 0.13775919377803802, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9565219283103943, "rewards/repeat_soft/std": 0.016908472403883934, "rewards/judge_quality/mean": 0.6737500429153442, "rewards/judge_quality/std": 0.263435423374176, "rewards/total_composite/mean": 0.7228672504425049, "rewards/total_composite/std": 0.13182733952999115, "reward": 0.7228672504425049, "reward_std": 0.13182730972766876, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10597249120473862, "sampling/sampling_logp_difference/max": 0.9990348815917969, "sampling/importance_sampling_ratio/min": 0.36823469400405884, "sampling/importance_sampling_ratio/mean": 1.0141311883926392, "sampling/importance_sampling_ratio/max": 1.60276460647583, "entropy": 0.8347270488739014, "clip_ratio/low_mean": 0.07963878381997347, "clip_ratio/low_min": 0.07963878381997347, "clip_ratio/high_mean": 0.03393665188923478, "clip_ratio/high_max": 0.03393665188923478, "clip_ratio/region_mean": 0.11357543570920825, "reward_total_mean": 0.7228672504425049, "reward_meter_mean": 0.8964908719062805, "reward_meter_std": 0.13775919377803802, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9565219283103943, "reward_repeat_soft_std": 0.016908472403883934, "reward_judge_quality_mean": 0.6737500429153442, "reward_judge_quality_std": 0.263435423374176, "reward_total_composite_mean": 0.7228672504425049, "reward_total_composite_std": 0.13182733952999115} {"timestamp_utc": "2026-04-13T11:40:54Z", "mode": "train", "global_step": 1758, "epoch": 0.17659467604218984, "loss": 0.0337, "grad_norm": 16.53756332397461, "learning_rate": 4.675757575757576e-06, "num_tokens": 3132882.0, "completions/mean_length": 35.5, "completions/min_length": 28.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.5, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.8374530076980591, "rewards/meter/std": 0.16281481087207794, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8819432258605957, "rewards/repeat_soft/std": 0.10562364757061005, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.6346421241760254, "rewards/total_composite/std": 0.1582137793302536, "reward": 0.6346421241760254, "reward_std": 0.1582137942314148, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11359891295433044, "sampling/sampling_logp_difference/max": 1.0051372051239014, "sampling/importance_sampling_ratio/min": 0.3659944236278534, "sampling/importance_sampling_ratio/mean": 1.0092700719833374, "sampling/importance_sampling_ratio/max": 1.958479642868042, "entropy": 0.6273365765810013, "clip_ratio/low_mean": 0.05642253835685551, "clip_ratio/low_min": 0.05642253835685551, "clip_ratio/high_mean": 0.020551801659166813, "clip_ratio/high_max": 0.020551801659166813, "clip_ratio/region_mean": 0.07697434001602232, "reward_total_mean": 0.6346421241760254, "reward_meter_mean": 0.8374530076980591, "reward_meter_std": 0.16281481087207794, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8819432258605957, "reward_repeat_soft_std": 0.10562364757061005, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.6346421241760254, "reward_total_composite_std": 0.1582137793302536} {"timestamp_utc": "2026-04-13T11:41:02Z", "mode": "train", "global_step": 1759, "epoch": 0.17669512807634355, "loss": 0.0614, "grad_norm": 5.83065938949585, "learning_rate": 4.6727272727272735e-06, "num_tokens": 3135179.0, "completions/mean_length": 116.125, "completions/min_length": 100.0, "completions/max_length": 164.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 116.125, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 164.0, "rewards/meter/mean": 0.8520817160606384, "rewards/meter/std": 0.22800517082214355, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7942557334899902, "rewards/repeat_soft/std": 0.10239209234714508, "rewards/judge_quality/mean": 0.6024999618530273, "rewards/judge_quality/std": 0.25217628479003906, "rewards/total_composite/mean": 0.5905674695968628, "rewards/total_composite/std": 0.15837174654006958, "reward": 0.5905674695968628, "reward_std": 0.15837174654006958, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0862930417060852, "sampling/sampling_logp_difference/max": 4.108579635620117, "sampling/importance_sampling_ratio/min": 0.01643109694123268, "sampling/importance_sampling_ratio/mean": 1.0071531534194946, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5726190023124218, "clip_ratio/low_mean": 0.055389934219419956, "clip_ratio/low_min": 0.055389934219419956, "clip_ratio/high_mean": 0.019607843831181526, "clip_ratio/high_max": 0.019607843831181526, "clip_ratio/region_mean": 0.07499777805060148, "reward_total_mean": 0.5905674695968628, "reward_meter_mean": 0.8520817160606384, "reward_meter_std": 0.22800517082214355, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7942557334899902, "reward_repeat_soft_std": 0.10239209234714508, "reward_judge_quality_mean": 0.6024999618530273, "reward_judge_quality_std": 0.25217628479003906, "reward_total_composite_mean": 0.5905674695968628, "reward_total_composite_std": 0.15837174654006958} {"timestamp_utc": "2026-04-13T11:41:08Z", "mode": "train", "global_step": 1760, "epoch": 0.17679558011049723, "loss": 0.0066, "grad_norm": 9.593058586120605, "learning_rate": 4.66969696969697e-06, "num_tokens": 3136704.0, "completions/mean_length": 33.625, "completions/min_length": 31.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.625, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9631706476211548, "rewards/meter/std": 0.012758648954331875, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8936579823493958, "rewards/repeat_soft/std": 0.032854970544576645, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.6759107112884521, "rewards/total_composite/std": 0.14938969910144806, "reward": 0.6759107112884521, "reward_std": 0.14938968420028687, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07251586765050888, "sampling/sampling_logp_difference/max": 1.6966581344604492, "sampling/importance_sampling_ratio/min": 0.18329504132270813, "sampling/importance_sampling_ratio/mean": 0.9984449744224548, "sampling/importance_sampling_ratio/max": 1.776847243309021, "entropy": 0.3891554996371269, "clip_ratio/low_mean": 0.04817867837846279, "clip_ratio/low_min": 0.04817867837846279, "clip_ratio/high_mean": 0.040088385343551636, "clip_ratio/high_max": 0.040088385343551636, "clip_ratio/region_mean": 0.08826706372201443, "reward_total_mean": 0.6759107112884521, "reward_meter_mean": 0.9631706476211548, "reward_meter_std": 0.012758648954331875, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8936579823493958, "reward_repeat_soft_std": 0.032854970544576645, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.6759107112884521, "reward_total_composite_std": 0.14938969910144806} {"timestamp_utc": "2026-04-13T11:41:16Z", "mode": "train", "global_step": 1761, "epoch": 0.17689603214465094, "loss": 0.0463, "grad_norm": 5.957404136657715, "learning_rate": 4.666666666666667e-06, "num_tokens": 3139106.0, "completions/mean_length": 105.25, "completions/min_length": 96.0, "completions/max_length": 120.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 105.25, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 120.0, "rewards/meter/mean": 0.7072615027427673, "rewards/meter/std": 0.19589649140834808, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7279385328292847, "rewards/repeat_soft/std": 0.06776217371225357, "rewards/judge_quality/mean": 0.7950000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.6447337865829468, "rewards/total_composite/std": 0.17704874277114868, "reward": 0.6447337865829468, "reward_std": 0.17704874277114868, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0715804472565651, "sampling/sampling_logp_difference/max": 2.039142370223999, "sampling/importance_sampling_ratio/min": 0.1301402747631073, "sampling/importance_sampling_ratio/mean": 1.0107367038726807, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.38171785324811935, "clip_ratio/low_mean": 0.03635486401617527, "clip_ratio/low_min": 0.03635486401617527, "clip_ratio/high_mean": 0.040356777142733335, "clip_ratio/high_max": 0.040356777142733335, "clip_ratio/region_mean": 0.0767116411589086, "reward_total_mean": 0.6447337865829468, "reward_meter_mean": 0.7072615027427673, "reward_meter_std": 0.19589649140834808, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7279385328292847, "reward_repeat_soft_std": 0.06776217371225357, "reward_judge_quality_mean": 0.7950000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.6447337865829468, "reward_total_composite_std": 0.17704874277114868} {"timestamp_utc": "2026-04-13T11:41:28Z", "mode": "train", "global_step": 1762, "epoch": 0.17699648417880462, "loss": -0.11, "grad_norm": 2.159605026245117, "learning_rate": 4.663636363636364e-06, "num_tokens": 3140556.0, "completions/mean_length": 95.25, "completions/min_length": 29.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 35.71428680419922, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.6219979524612427, "rewards/meter/std": 0.4170207977294922, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.933821439743042, "rewards/repeat_soft/std": 0.043045803904533386, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.4652155339717865, "rewards/total_composite/std": 0.20921353995800018, "reward": 0.4652155339717865, "reward_std": 0.209213525056839, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0978228896856308, "sampling/sampling_logp_difference/max": 1.2196571826934814, "sampling/importance_sampling_ratio/min": 0.2953313887119293, "sampling/importance_sampling_ratio/mean": 1.0135092735290527, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5758384056389332, "clip_ratio/low_mean": 0.016607505502179265, "clip_ratio/low_min": 0.016607505502179265, "clip_ratio/high_mean": 0.08218754455447197, "clip_ratio/high_max": 0.08218754455447197, "clip_ratio/region_mean": 0.09879505005665123, "reward_total_mean": 0.4652155339717865, "reward_meter_mean": 0.6219979524612427, "reward_meter_std": 0.4170207977294922, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.933821439743042, "reward_repeat_soft_std": 0.043045803904533386, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.4652155339717865, "reward_total_composite_std": 0.20921353995800018} {"timestamp_utc": "2026-04-13T11:41:36Z", "mode": "train", "global_step": 1763, "epoch": 0.1770969362129583, "loss": -0.0068, "grad_norm": 5.532903671264648, "learning_rate": 4.660606060606061e-06, "num_tokens": 3142956.0, "completions/mean_length": 119.0, "completions/min_length": 103.0, "completions/max_length": 131.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 119.0, "completions/min_terminated_length": 103.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.5499364733695984, "rewards/meter/std": 0.3466138243675232, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7667504549026489, "rewards/repeat_soft/std": 0.045128677040338516, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.4173145890235901, "rewards/total_composite/std": 0.09702639281749725, "reward": 0.4173145890235901, "reward_std": 0.09702640026807785, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07352930307388306, "sampling/sampling_logp_difference/max": 1.6493425369262695, "sampling/importance_sampling_ratio/min": 0.21601814031600952, "sampling/importance_sampling_ratio/mean": 1.007866382598877, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.46876227855682373, "clip_ratio/low_mean": 0.026812857948243618, "clip_ratio/low_min": 0.026812857948243618, "clip_ratio/high_mean": 0.05130196502432227, "clip_ratio/high_max": 0.05130196502432227, "clip_ratio/region_mean": 0.07811482297256589, "reward_total_mean": 0.4173145890235901, "reward_meter_mean": 0.5499364733695984, "reward_meter_std": 0.3466138243675232, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7667504549026489, "reward_repeat_soft_std": 0.045128677040338516, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.4173145890235901, "reward_total_composite_std": 0.09702639281749725} {"timestamp_utc": "2026-04-13T11:41:42Z", "mode": "train", "global_step": 1764, "epoch": 0.177197388247112, "loss": 0.0115, "grad_norm": 12.079869270324707, "learning_rate": 4.657575757575758e-06, "num_tokens": 3144733.0, "completions/mean_length": 54.125, "completions/min_length": 47.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.125, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.8532311916351318, "rewards/meter/std": 0.2813609540462494, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9801090955734253, "rewards/repeat_soft/std": 0.027251332998275757, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.10507651418447495, "rewards/total_composite/mean": 0.6062156558036804, "rewards/total_composite/std": 0.11235784739255905, "reward": 0.6062156558036804, "reward_std": 0.11235783994197845, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13048695027828217, "sampling/sampling_logp_difference/max": 1.5955047607421875, "sampling/importance_sampling_ratio/min": 0.20280614495277405, "sampling/importance_sampling_ratio/mean": 1.0123000144958496, "sampling/importance_sampling_ratio/max": 1.858381748199463, "entropy": 0.8582986369729042, "clip_ratio/low_mean": 0.048888481222093105, "clip_ratio/low_min": 0.048888481222093105, "clip_ratio/high_mean": 0.09009198471903801, "clip_ratio/high_max": 0.09009198471903801, "clip_ratio/region_mean": 0.13898046594113111, "reward_total_mean": 0.6062156558036804, "reward_meter_mean": 0.8532311916351318, "reward_meter_std": 0.2813609540462494, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9801090955734253, "reward_repeat_soft_std": 0.027251332998275757, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.10507651418447495, "reward_total_composite_mean": 0.6062156558036804, "reward_total_composite_std": 0.11235784739255905} {"timestamp_utc": "2026-04-13T11:41:54Z", "mode": "train", "global_step": 1765, "epoch": 0.1772978402812657, "loss": -0.1219, "grad_norm": 1.6977033615112305, "learning_rate": 4.654545454545455e-06, "num_tokens": 3146297.0, "completions/mean_length": 234.5, "completions/min_length": 57.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 68.0, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.6177030801773071, "rewards/meter/std": 0.39913487434387207, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.17817415297031403, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.8289181590080261, "rewards/repeat_soft/std": 0.1014774739742279, "rewards/judge_quality/mean": 0.25999999046325684, "rewards/judge_quality/std": 0.18314708769321442, "rewards/total_composite/mean": 0.2794647812843323, "rewards/total_composite/std": 0.2390124350786209, "reward": 0.2794647812843323, "reward_std": 0.2390124350786209, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09115440398454666, "sampling/sampling_logp_difference/max": 1.0756086111068726, "sampling/importance_sampling_ratio/min": 0.3410901129245758, "sampling/importance_sampling_ratio/mean": 1.0201588869094849, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.42579662427306175, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.05242840852588415, "clip_ratio/high_max": 0.05242840852588415, "clip_ratio/region_mean": 0.05242840852588415, "reward_total_mean": 0.2794647812843323, "reward_meter_mean": 0.6177030801773071, "reward_meter_std": 0.39913487434387207, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.17817415297031403, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.8289181590080261, "reward_repeat_soft_std": 0.1014774739742279, "reward_judge_quality_mean": 0.25999999046325684, "reward_judge_quality_std": 0.18314708769321442, "reward_total_composite_mean": 0.2794647812843323, "reward_total_composite_std": 0.2390124350786209} {"timestamp_utc": "2026-04-13T11:42:01Z", "mode": "train", "global_step": 1766, "epoch": 0.1773982923154194, "loss": -0.0363, "grad_norm": 12.41996955871582, "learning_rate": 4.651515151515152e-06, "num_tokens": 3147997.0, "completions/mean_length": 64.5, "completions/min_length": 40.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.5, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.7062945365905762, "rewards/meter/std": 0.3823467791080475, "rewards/count_adherence/mean": 0.7083333730697632, "rewards/count_adherence/std": 0.11785111576318741, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9083278179168701, "rewards/repeat_soft/std": 0.05187005177140236, "rewards/judge_quality/mean": 0.5950000286102295, "rewards/judge_quality/std": 0.19820626080036163, "rewards/total_composite/mean": 0.5370651483535767, "rewards/total_composite/std": 0.14727263152599335, "reward": 0.5370651483535767, "reward_std": 0.14727263152599335, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14444056153297424, "sampling/sampling_logp_difference/max": 2.5635123252868652, "sampling/importance_sampling_ratio/min": 0.07703369855880737, "sampling/importance_sampling_ratio/mean": 0.9915181398391724, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5885419137775898, "clip_ratio/low_mean": 0.05729159014299512, "clip_ratio/low_min": 0.05729159014299512, "clip_ratio/high_mean": 0.05323988106101751, "clip_ratio/high_max": 0.05323988106101751, "clip_ratio/region_mean": 0.11053147120401263, "reward_total_mean": 0.5370651483535767, "reward_meter_mean": 0.7062945365905762, "reward_meter_std": 0.3823467791080475, "reward_count_adherence_mean": 0.7083333730697632, "reward_count_adherence_std": 0.11785111576318741, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9083278179168701, "reward_repeat_soft_std": 0.05187005177140236, "reward_judge_quality_mean": 0.5950000286102295, "reward_judge_quality_std": 0.19820626080036163, "reward_total_composite_mean": 0.5370651483535767, "reward_total_composite_std": 0.14727263152599335} {"timestamp_utc": "2026-04-13T11:42:07Z", "mode": "train", "global_step": 1767, "epoch": 0.17749874434957308, "loss": -0.0102, "grad_norm": 13.42194938659668, "learning_rate": 4.648484848484849e-06, "num_tokens": 3149559.0, "completions/mean_length": 32.25, "completions/min_length": 31.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.25, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.958257794380188, "rewards/meter/std": 0.005651597864925861, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7578088045120239, "rewards/repeat_soft/std": 0.05588805302977562, "rewards/judge_quality/mean": 0.7950000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.8086973428726196, "rewards/total_composite/std": 0.1457444578409195, "reward": 0.8086973428726196, "reward_std": 0.1457444280385971, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.039556313306093216, "sampling/sampling_logp_difference/max": 1.269411563873291, "sampling/importance_sampling_ratio/min": 0.2809969186782837, "sampling/importance_sampling_ratio/mean": 0.9951732754707336, "sampling/importance_sampling_ratio/max": 1.454006552696228, "entropy": 0.1566576324403286, "clip_ratio/low_mean": 0.008064515888690948, "clip_ratio/low_min": 0.008064515888690948, "clip_ratio/high_mean": 0.022741199005395174, "clip_ratio/high_max": 0.022741199005395174, "clip_ratio/region_mean": 0.030805714894086123, "reward_total_mean": 0.8086973428726196, "reward_meter_mean": 0.958257794380188, "reward_meter_std": 0.005651597864925861, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7578088045120239, "reward_repeat_soft_std": 0.05588805302977562, "reward_judge_quality_mean": 0.7950000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.8086973428726196, "reward_total_composite_std": 0.1457444578409195} {"timestamp_utc": "2026-04-13T11:42:13Z", "mode": "train", "global_step": 1768, "epoch": 0.17759919638372676, "loss": 0.0193, "grad_norm": 12.585282325744629, "learning_rate": 4.645454545454545e-06, "num_tokens": 3151470.0, "completions/mean_length": 48.875, "completions/min_length": 46.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.875, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.8644855618476868, "rewards/meter/std": 0.3265610635280609, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9649581909179688, "rewards/repeat_soft/std": 0.03800182044506073, "rewards/judge_quality/mean": 0.4424999952316284, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.5927733778953552, "rewards/total_composite/std": 0.09315738081932068, "reward": 0.5927733778953552, "reward_std": 0.09315737336874008, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12805289030075073, "sampling/sampling_logp_difference/max": 1.685804843902588, "sampling/importance_sampling_ratio/min": 0.1852952390909195, "sampling/importance_sampling_ratio/mean": 1.0057452917099, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7312658205628395, "clip_ratio/low_mean": 0.0076530613005161285, "clip_ratio/low_min": 0.0076530613005161285, "clip_ratio/high_mean": 0.11520470632240176, "clip_ratio/high_max": 0.11520470632240176, "clip_ratio/region_mean": 0.12285776762291789, "reward_total_mean": 0.5927733778953552, "reward_meter_mean": 0.8644855618476868, "reward_meter_std": 0.3265610635280609, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9649581909179688, "reward_repeat_soft_std": 0.03800182044506073, "reward_judge_quality_mean": 0.4424999952316284, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.5927733778953552, "reward_total_composite_std": 0.09315738081932068} {"timestamp_utc": "2026-04-13T11:42:21Z", "mode": "train", "global_step": 1769, "epoch": 0.17769964841788047, "loss": 0.0357, "grad_norm": 4.545206069946289, "learning_rate": 4.642424242424243e-06, "num_tokens": 3154059.0, "completions/mean_length": 143.625, "completions/min_length": 119.0, "completions/max_length": 167.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 143.625, "completions/min_terminated_length": 119.0, "completions/max_terminated_length": 167.0, "rewards/meter/mean": 0.8274730443954468, "rewards/meter/std": 0.3358956575393677, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7043778896331787, "rewards/repeat_soft/std": 0.10080349445343018, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.2834985554218292, "rewards/total_composite/mean": 0.5214362144470215, "rewards/total_composite/std": 0.200816810131073, "reward": 0.5214362144470215, "reward_std": 0.200816810131073, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07771973311901093, "sampling/sampling_logp_difference/max": 2.0672922134399414, "sampling/importance_sampling_ratio/min": 0.12652792036533356, "sampling/importance_sampling_ratio/mean": 1.002564787864685, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.42923061549663544, "clip_ratio/low_mean": 0.029658550396561623, "clip_ratio/low_min": 0.029658550396561623, "clip_ratio/high_mean": 0.0478785103186965, "clip_ratio/high_max": 0.0478785103186965, "clip_ratio/region_mean": 0.07753706071525812, "reward_total_mean": 0.5214362144470215, "reward_meter_mean": 0.8274730443954468, "reward_meter_std": 0.3358956575393677, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7043778896331787, "reward_repeat_soft_std": 0.10080349445343018, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.2834985554218292, "reward_total_composite_mean": 0.5214362144470215, "reward_total_composite_std": 0.200816810131073} {"timestamp_utc": "2026-04-13T11:42:28Z", "mode": "train", "global_step": 1770, "epoch": 0.17780010045203415, "loss": 0.086, "grad_norm": 10.989616394042969, "learning_rate": 4.63939393939394e-06, "num_tokens": 3156199.0, "completions/mean_length": 85.5, "completions/min_length": 68.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.5, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.9170221090316772, "rewards/meter/std": 0.15822051465511322, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7349417209625244, "rewards/repeat_soft/std": 0.14873848855495453, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.49723243713378906, "rewards/total_composite/std": 0.04901236668229103, "reward": 0.49723243713378906, "reward_std": 0.04901236668229103, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09723883122205734, "sampling/sampling_logp_difference/max": 1.4950122833251953, "sampling/importance_sampling_ratio/min": 0.2242458611726761, "sampling/importance_sampling_ratio/mean": 1.004080891609192, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6100311912596226, "clip_ratio/low_mean": 0.029673343524336815, "clip_ratio/low_min": 0.029673343524336815, "clip_ratio/high_mean": 0.06719653774052858, "clip_ratio/high_max": 0.06719653774052858, "clip_ratio/region_mean": 0.0968698812648654, "reward_total_mean": 0.49723243713378906, "reward_meter_mean": 0.9170221090316772, "reward_meter_std": 0.15822051465511322, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7349417209625244, "reward_repeat_soft_std": 0.14873848855495453, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.49723243713378906, "reward_total_composite_std": 0.04901236668229103} {"timestamp_utc": "2026-04-13T11:42:37Z", "mode": "train", "global_step": 1771, "epoch": 0.17790055248618786, "loss": -0.0555, "grad_norm": 8.31665325164795, "learning_rate": 4.636363636363636e-06, "num_tokens": 3159009.0, "completions/mean_length": 165.25, "completions/min_length": 124.0, "completions/max_length": 195.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 165.25, "completions/min_terminated_length": 124.0, "completions/max_terminated_length": 195.0, "rewards/meter/mean": 0.9113659858703613, "rewards/meter/std": 0.203917995095253, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8120887875556946, "rewards/repeat_soft/std": 0.09801527857780457, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5306162238121033, "rewards/total_composite/std": 0.0546850860118866, "reward": 0.5306162238121033, "reward_std": 0.054685078561306, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0938008725643158, "sampling/sampling_logp_difference/max": 2.30631160736084, "sampling/importance_sampling_ratio/min": 0.0996280387043953, "sampling/importance_sampling_ratio/mean": 1.0084292888641357, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.559680700302124, "clip_ratio/low_mean": 0.019789081532508135, "clip_ratio/low_min": 0.019789081532508135, "clip_ratio/high_mean": 0.06842813175171614, "clip_ratio/high_max": 0.06842813175171614, "clip_ratio/region_mean": 0.08821721328422427, "reward_total_mean": 0.5306162238121033, "reward_meter_mean": 0.9113659858703613, "reward_meter_std": 0.203917995095253, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8120887875556946, "reward_repeat_soft_std": 0.09801527857780457, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5306162238121033, "reward_total_composite_std": 0.0546850860118866} {"timestamp_utc": "2026-04-13T11:42:44Z", "mode": "train", "global_step": 1772, "epoch": 0.17800100452034154, "loss": 0.0047, "grad_norm": 7.811924457550049, "learning_rate": 4.633333333333334e-06, "num_tokens": 3161518.0, "completions/mean_length": 121.625, "completions/min_length": 102.0, "completions/max_length": 155.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 121.625, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 155.0, "rewards/meter/mean": 0.4056808650493622, "rewards/meter/std": 0.32659587264060974, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.819111704826355, "rewards/repeat_soft/std": 0.07080240547657013, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4286176264286041, "rewards/total_composite/std": 0.08320928364992142, "reward": 0.4286176264286041, "reward_std": 0.08320929110050201, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1050177738070488, "sampling/sampling_logp_difference/max": 3.019057273864746, "sampling/importance_sampling_ratio/min": 0.04884724318981171, "sampling/importance_sampling_ratio/mean": 1.0052545070648193, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6332890838384628, "clip_ratio/low_mean": 0.04819859517738223, "clip_ratio/low_min": 0.04819859517738223, "clip_ratio/high_mean": 0.03790410188958049, "clip_ratio/high_max": 0.03790410188958049, "clip_ratio/region_mean": 0.08610269706696272, "reward_total_mean": 0.4286176264286041, "reward_meter_mean": 0.4056808650493622, "reward_meter_std": 0.32659587264060974, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.819111704826355, "reward_repeat_soft_std": 0.07080240547657013, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4286176264286041, "reward_total_composite_std": 0.08320928364992142} {"timestamp_utc": "2026-04-13T11:42:52Z", "mode": "train", "global_step": 1773, "epoch": 0.17810145655449522, "loss": 0.0723, "grad_norm": 6.276249408721924, "learning_rate": 4.630303030303031e-06, "num_tokens": 3163854.0, "completions/mean_length": 120.0, "completions/min_length": 104.0, "completions/max_length": 135.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 120.0, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 135.0, "rewards/meter/mean": 0.4395008683204651, "rewards/meter/std": 0.27812936902046204, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.7932449579238892, "rewards/repeat_soft/std": 0.0670030266046524, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.343326210975647, "rewards/total_composite/std": 0.15900444984436035, "reward": 0.343326210975647, "reward_std": 0.15900444984436035, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09043148159980774, "sampling/sampling_logp_difference/max": 1.6271238327026367, "sampling/importance_sampling_ratio/min": 0.19649390876293182, "sampling/importance_sampling_ratio/mean": 1.0063484907150269, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6767326816916466, "clip_ratio/low_mean": 0.045293652918189764, "clip_ratio/low_min": 0.045293652918189764, "clip_ratio/high_mean": 0.04355377238243818, "clip_ratio/high_max": 0.04355377238243818, "clip_ratio/region_mean": 0.08884742530062795, "reward_total_mean": 0.343326210975647, "reward_meter_mean": 0.4395008683204651, "reward_meter_std": 0.27812936902046204, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.7932449579238892, "reward_repeat_soft_std": 0.0670030266046524, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.343326210975647, "reward_total_composite_std": 0.15900444984436035} {"timestamp_utc": "2026-04-13T11:42:58Z", "mode": "train", "global_step": 1774, "epoch": 0.17820190858864893, "loss": 0.2022, "grad_norm": 21.477895736694336, "learning_rate": 4.627272727272727e-06, "num_tokens": 3165287.0, "completions/mean_length": 20.125, "completions/min_length": 12.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 20.125, "completions/min_terminated_length": 12.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.3956363797187805, "rewards/meter/std": 0.359622985124588, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9515777826309204, "rewards/repeat_soft/std": 0.02880319207906723, "rewards/judge_quality/mean": 0.3799999952316284, "rewards/judge_quality/std": 0.11501552164554596, "rewards/total_composite/mean": 0.4245898127555847, "rewards/total_composite/std": 0.13764996826648712, "reward": 0.4245898127555847, "reward_std": 0.13764995336532593, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15270955860614777, "sampling/sampling_logp_difference/max": 1.2393665313720703, "sampling/importance_sampling_ratio/min": 0.28956758975982666, "sampling/importance_sampling_ratio/mean": 1.0259755849838257, "sampling/importance_sampling_ratio/max": 1.914949655532837, "entropy": 1.2450411096215248, "clip_ratio/low_mean": 0.09484980208799243, "clip_ratio/low_min": 0.09484980208799243, "clip_ratio/high_mean": 0.056746033020317554, "clip_ratio/high_max": 0.056746033020317554, "clip_ratio/region_mean": 0.15159583510830998, "reward_total_mean": 0.4245898127555847, "reward_meter_mean": 0.3956363797187805, "reward_meter_std": 0.359622985124588, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9515777826309204, "reward_repeat_soft_std": 0.02880319207906723, "reward_judge_quality_mean": 0.3799999952316284, "reward_judge_quality_std": 0.11501552164554596, "reward_total_composite_mean": 0.4245898127555847, "reward_total_composite_std": 0.13764996826648712} {"timestamp_utc": "2026-04-13T11:43:05Z", "mode": "train", "global_step": 1775, "epoch": 0.1783023606228026, "loss": 0.0943, "grad_norm": 7.26712703704834, "learning_rate": 4.6242424242424245e-06, "num_tokens": 3167474.0, "completions/mean_length": 101.375, "completions/min_length": 82.0, "completions/max_length": 122.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 101.375, "completions/min_terminated_length": 82.0, "completions/max_terminated_length": 122.0, "rewards/meter/mean": 0.5810152888298035, "rewards/meter/std": 0.2275850921869278, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8646507263183594, "rewards/repeat_soft/std": 0.06669933348894119, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.42982417345046997, "rewards/total_composite/std": 0.05832752585411072, "reward": 0.42982417345046997, "reward_std": 0.05832751840353012, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09182518720626831, "sampling/sampling_logp_difference/max": 2.051914930343628, "sampling/importance_sampling_ratio/min": 0.12848861515522003, "sampling/importance_sampling_ratio/mean": 1.00731360912323, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5825636647641659, "clip_ratio/low_mean": 0.05231765937060118, "clip_ratio/low_min": 0.05231765937060118, "clip_ratio/high_mean": 0.04964388348162174, "clip_ratio/high_max": 0.04964388348162174, "clip_ratio/region_mean": 0.10196154285222292, "reward_total_mean": 0.42982417345046997, "reward_meter_mean": 0.5810152888298035, "reward_meter_std": 0.2275850921869278, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8646507263183594, "reward_repeat_soft_std": 0.06669933348894119, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.42982417345046997, "reward_total_composite_std": 0.05832752585411072} {"timestamp_utc": "2026-04-13T11:43:11Z", "mode": "train", "global_step": 1776, "epoch": 0.17840281265695632, "loss": 0.0004, "grad_norm": 19.19764518737793, "learning_rate": 4.621212121212122e-06, "num_tokens": 3169086.0, "completions/mean_length": 22.5, "completions/min_length": 19.0, "completions/max_length": 27.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.5, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.7644567489624023, "rewards/meter/std": 0.37200382351875305, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.95320725440979, "rewards/repeat_soft/std": 0.022005343809723854, "rewards/judge_quality/mean": 0.38874998688697815, "rewards/judge_quality/std": 0.08675704896450043, "rewards/total_composite/mean": 0.5397839546203613, "rewards/total_composite/std": 0.11333271116018295, "reward": 0.5397839546203613, "reward_std": 0.11333271116018295, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14350047707557678, "sampling/sampling_logp_difference/max": 1.3546104431152344, "sampling/importance_sampling_ratio/min": 0.25804781913757324, "sampling/importance_sampling_ratio/mean": 1.0243251323699951, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0798679664731026, "clip_ratio/low_mean": 0.02336956560611725, "clip_ratio/low_min": 0.02336956560611725, "clip_ratio/high_mean": 0.10425302013754845, "clip_ratio/high_max": 0.10425302013754845, "clip_ratio/region_mean": 0.1276225857436657, "reward_total_mean": 0.5397839546203613, "reward_meter_mean": 0.7644567489624023, "reward_meter_std": 0.37200382351875305, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.95320725440979, "reward_repeat_soft_std": 0.022005343809723854, "reward_judge_quality_mean": 0.38874998688697815, "reward_judge_quality_std": 0.08675704896450043, "reward_total_composite_mean": 0.5397839546203613, "reward_total_composite_std": 0.11333271116018295} {"timestamp_utc": "2026-04-13T11:43:18Z", "mode": "train", "global_step": 1777, "epoch": 0.17850326469111, "loss": 0.0508, "grad_norm": 14.726597785949707, "learning_rate": 4.618181818181818e-06, "num_tokens": 3170659.0, "completions/mean_length": 34.625, "completions/min_length": 32.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.625, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.8540171980857849, "rewards/meter/std": 0.294059693813324, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8245512843132019, "rewards/repeat_soft/std": 0.07131902873516083, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.5434920787811279, "rewards/total_composite/std": 0.0831054225564003, "reward": 0.5434920787811279, "reward_std": 0.0831054225564003, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07951752841472626, "sampling/sampling_logp_difference/max": 1.0437231063842773, "sampling/importance_sampling_ratio/min": 0.3521411716938019, "sampling/importance_sampling_ratio/mean": 0.9965921640396118, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4809534139931202, "clip_ratio/low_mean": 0.023838141933083534, "clip_ratio/low_min": 0.023838141933083534, "clip_ratio/high_mean": 0.05456614587455988, "clip_ratio/high_max": 0.05456614587455988, "clip_ratio/region_mean": 0.07840428780764341, "reward_total_mean": 0.5434920787811279, "reward_meter_mean": 0.8540171980857849, "reward_meter_std": 0.294059693813324, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8245512843132019, "reward_repeat_soft_std": 0.07131902873516083, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.5434920787811279, "reward_total_composite_std": 0.0831054225564003} {"timestamp_utc": "2026-04-13T11:43:24Z", "mode": "train", "global_step": 1778, "epoch": 0.17860371672526368, "loss": 0.0254, "grad_norm": 11.550619125366211, "learning_rate": 4.615151515151515e-06, "num_tokens": 3172276.0, "completions/mean_length": 35.125, "completions/min_length": 33.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.125, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9001180529594421, "rewards/meter/std": 0.1616150140762329, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.76820969581604, "rewards/repeat_soft/std": 0.07499588280916214, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5609637498855591, "rewards/total_composite/std": 0.045429080724716187, "reward": 0.5609637498855591, "reward_std": 0.04542907327413559, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04459671676158905, "sampling/sampling_logp_difference/max": 0.7482872009277344, "sampling/importance_sampling_ratio/min": 0.4731763005256653, "sampling/importance_sampling_ratio/mean": 0.9990738034248352, "sampling/importance_sampling_ratio/max": 1.7169886827468872, "entropy": 0.27748790569603443, "clip_ratio/low_mean": 0.0033783784601837397, "clip_ratio/low_min": 0.0033783784601837397, "clip_ratio/high_mean": 0.043360243551433086, "clip_ratio/high_max": 0.043360243551433086, "clip_ratio/region_mean": 0.046738622011616826, "reward_total_mean": 0.5609637498855591, "reward_meter_mean": 0.9001180529594421, "reward_meter_std": 0.1616150140762329, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.76820969581604, "reward_repeat_soft_std": 0.07499588280916214, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5609637498855591, "reward_total_composite_std": 0.045429080724716187} {"timestamp_utc": "2026-04-13T11:43:34Z", "mode": "train", "global_step": 1779, "epoch": 0.1787041687594174, "loss": -0.0005, "grad_norm": 3.2185769081115723, "learning_rate": 4.612121212121212e-06, "num_tokens": 3175499.0, "completions/mean_length": 209.875, "completions/min_length": 169.0, "completions/max_length": 240.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 209.875, "completions/min_terminated_length": 169.0, "completions/max_terminated_length": 240.0, "rewards/meter/mean": 0.8809875845909119, "rewards/meter/std": 0.14933769404888153, "rewards/count_adherence/mean": 0.7083333730697632, "rewards/count_adherence/std": 0.07715165615081787, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6478923559188843, "rewards/repeat_soft/std": 0.08954145014286041, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.45875000953674316, "rewards/total_composite/std": 0.06916475296020508, "reward": 0.45875000953674316, "reward_std": 0.06916474550962448, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05661586672067642, "sampling/sampling_logp_difference/max": 1.8036327362060547, "sampling/importance_sampling_ratio/min": 0.1646994948387146, "sampling/importance_sampling_ratio/mean": 1.0127066373825073, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3152757529169321, "clip_ratio/low_mean": 0.02137986244633794, "clip_ratio/low_min": 0.02137986244633794, "clip_ratio/high_mean": 0.025456495000980794, "clip_ratio/high_max": 0.025456495000980794, "clip_ratio/region_mean": 0.04683635744731873, "reward_total_mean": 0.45875000953674316, "reward_meter_mean": 0.8809875845909119, "reward_meter_std": 0.14933769404888153, "reward_count_adherence_mean": 0.7083333730697632, "reward_count_adherence_std": 0.07715165615081787, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6478923559188843, "reward_repeat_soft_std": 0.08954145014286041, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.45875000953674316, "reward_total_composite_std": 0.06916475296020508} {"timestamp_utc": "2026-04-13T11:43:40Z", "mode": "train", "global_step": 1780, "epoch": 0.17880462079357107, "loss": 0.0483, "grad_norm": 8.797980308532715, "learning_rate": 4.60909090909091e-06, "num_tokens": 3177122.0, "completions/mean_length": 52.875, "completions/min_length": 42.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.875, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.8468008637428284, "rewards/meter/std": 0.22864817082881927, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9050402045249939, "rewards/repeat_soft/std": 0.03166838362812996, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.6148871183395386, "rewards/total_composite/std": 0.12613748013973236, "reward": 0.6148871183395386, "reward_std": 0.12613748013973236, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12379765510559082, "sampling/sampling_logp_difference/max": 1.970367431640625, "sampling/importance_sampling_ratio/min": 0.13940562307834625, "sampling/importance_sampling_ratio/mean": 1.0071300268173218, "sampling/importance_sampling_ratio/max": 1.9179096221923828, "entropy": 0.7289240285754204, "clip_ratio/low_mean": 0.08008087985217571, "clip_ratio/low_min": 0.08008087985217571, "clip_ratio/high_mean": 0.026988636702299118, "clip_ratio/high_max": 0.026988636702299118, "clip_ratio/region_mean": 0.10706951655447483, "reward_total_mean": 0.6148871183395386, "reward_meter_mean": 0.8468008637428284, "reward_meter_std": 0.22864817082881927, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9050402045249939, "reward_repeat_soft_std": 0.03166838362812996, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.6148871183395386, "reward_total_composite_std": 0.12613748013973236} {"timestamp_utc": "2026-04-13T11:43:47Z", "mode": "train", "global_step": 1781, "epoch": 0.17890507282772475, "loss": 0.0441, "grad_norm": 7.625504493713379, "learning_rate": 4.606060606060606e-06, "num_tokens": 3179254.0, "completions/mean_length": 95.5, "completions/min_length": 78.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.5, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.6000702381134033, "rewards/meter/std": 0.3523312211036682, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.1178511381149292, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7699798941612244, "rewards/repeat_soft/std": 0.07029587030410767, "rewards/judge_quality/mean": 0.5487500429153442, "rewards/judge_quality/std": 0.22937415540218353, "rewards/total_composite/mean": 0.43533459305763245, "rewards/total_composite/std": 0.11670365929603577, "reward": 0.43533459305763245, "reward_std": 0.11670365929603577, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09412666410207748, "sampling/sampling_logp_difference/max": 1.4652800559997559, "sampling/importance_sampling_ratio/min": 0.23101328313350677, "sampling/importance_sampling_ratio/mean": 0.9994523525238037, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5815066248178482, "clip_ratio/low_mean": 0.028647522209212184, "clip_ratio/low_min": 0.028647522209212184, "clip_ratio/high_mean": 0.05417865049093962, "clip_ratio/high_max": 0.05417865049093962, "clip_ratio/region_mean": 0.0828261727001518, "reward_total_mean": 0.43533459305763245, "reward_meter_mean": 0.6000702381134033, "reward_meter_std": 0.3523312211036682, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.1178511381149292, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7699798941612244, "reward_repeat_soft_std": 0.07029587030410767, "reward_judge_quality_mean": 0.5487500429153442, "reward_judge_quality_std": 0.22937415540218353, "reward_total_composite_mean": 0.43533459305763245, "reward_total_composite_std": 0.11670365929603577} {"timestamp_utc": "2026-04-13T11:43:54Z", "mode": "train", "global_step": 1782, "epoch": 0.17900552486187846, "loss": 0.0737, "grad_norm": 16.86492347717285, "learning_rate": 4.603030303030304e-06, "num_tokens": 3180719.0, "completions/mean_length": 23.125, "completions/min_length": 17.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.125, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.6610732674598694, "rewards/meter/std": 0.45391368865966797, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9396507740020752, "rewards/repeat_soft/std": 0.04000112786889076, "rewards/judge_quality/mean": 0.26749998331069946, "rewards/judge_quality/std": 0.10375107079744339, "rewards/total_composite/mean": 0.46557289361953735, "rewards/total_composite/std": 0.10590554028749466, "reward": 0.46557289361953735, "reward_std": 0.10590554773807526, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19363300502300262, "sampling/sampling_logp_difference/max": 1.2690849304199219, "sampling/importance_sampling_ratio/min": 0.2810887396335602, "sampling/importance_sampling_ratio/mean": 1.028009295463562, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6260251477360725, "clip_ratio/low_mean": 0.11799331195652485, "clip_ratio/low_min": 0.11799331195652485, "clip_ratio/high_mean": 0.0705706044100225, "clip_ratio/high_max": 0.0705706044100225, "clip_ratio/region_mean": 0.18856391636654735, "reward_total_mean": 0.46557289361953735, "reward_meter_mean": 0.6610732674598694, "reward_meter_std": 0.45391368865966797, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9396507740020752, "reward_repeat_soft_std": 0.04000112786889076, "reward_judge_quality_mean": 0.26749998331069946, "reward_judge_quality_std": 0.10375107079744339, "reward_total_composite_mean": 0.46557289361953735, "reward_total_composite_std": 0.10590554028749466} {"timestamp_utc": "2026-04-13T11:44:01Z", "mode": "train", "global_step": 1783, "epoch": 0.17910597689603214, "loss": 0.0934, "grad_norm": 7.992717742919922, "learning_rate": 4.600000000000001e-06, "num_tokens": 3182501.0, "completions/mean_length": 57.75, "completions/min_length": 39.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.75, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.2404651790857315, "rewards/meter/std": 0.21960659325122833, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8838211297988892, "rewards/repeat_soft/std": 0.044775135815143585, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.3238157331943512, "rewards/total_composite/std": 0.06840462237596512, "reward": 0.3238157331943512, "reward_std": 0.06840461492538452, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10930705070495605, "sampling/sampling_logp_difference/max": 3.122370958328247, "sampling/importance_sampling_ratio/min": 0.044052597135305405, "sampling/importance_sampling_ratio/mean": 1.0260294675827026, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7410463839769363, "clip_ratio/low_mean": 0.02844047429971397, "clip_ratio/low_min": 0.02844047429971397, "clip_ratio/high_mean": 0.05527160316705704, "clip_ratio/high_max": 0.05527160316705704, "clip_ratio/region_mean": 0.083712077466771, "reward_total_mean": 0.3238157331943512, "reward_meter_mean": 0.2404651790857315, "reward_meter_std": 0.21960659325122833, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8838211297988892, "reward_repeat_soft_std": 0.044775135815143585, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.3238157331943512, "reward_total_composite_std": 0.06840462237596512} {"timestamp_utc": "2026-04-13T11:44:08Z", "mode": "train", "global_step": 1784, "epoch": 0.17920642893018585, "loss": 0.0604, "grad_norm": 11.344841957092285, "learning_rate": 4.596969696969697e-06, "num_tokens": 3184251.0, "completions/mean_length": 53.75, "completions/min_length": 41.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.75, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9381598234176636, "rewards/meter/std": 0.13511887192726135, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9147299528121948, "rewards/repeat_soft/std": 0.05088740214705467, "rewards/judge_quality/mean": 0.4724999964237213, "rewards/judge_quality/std": 0.1011011004447937, "rewards/total_composite/mean": 0.6177034378051758, "rewards/total_composite/std": 0.012627576477825642, "reward": 0.6177034378051758, "reward_std": 0.012627576477825642, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11097406595945358, "sampling/sampling_logp_difference/max": 1.2670319080352783, "sampling/importance_sampling_ratio/min": 0.2816663980484009, "sampling/importance_sampling_ratio/mean": 1.0102325677871704, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.744365319609642, "clip_ratio/low_mean": 0.04959837393835187, "clip_ratio/low_min": 0.04959837393835187, "clip_ratio/high_mean": 0.05398035328835249, "clip_ratio/high_max": 0.05398035328835249, "clip_ratio/region_mean": 0.10357872722670436, "reward_total_mean": 0.6177034378051758, "reward_meter_mean": 0.9381598234176636, "reward_meter_std": 0.13511887192726135, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9147299528121948, "reward_repeat_soft_std": 0.05088740214705467, "reward_judge_quality_mean": 0.4724999964237213, "reward_judge_quality_std": 0.1011011004447937, "reward_total_composite_mean": 0.6177034378051758, "reward_total_composite_std": 0.012627576477825642} {"timestamp_utc": "2026-04-13T11:44:16Z", "mode": "train", "global_step": 1785, "epoch": 0.17930688096433953, "loss": 0.0597, "grad_norm": 8.982309341430664, "learning_rate": 4.5939393939393945e-06, "num_tokens": 3187298.0, "completions/mean_length": 188.875, "completions/min_length": 170.0, "completions/max_length": 206.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 188.875, "completions/min_terminated_length": 170.0, "completions/max_terminated_length": 206.0, "rewards/meter/mean": 0.9501059055328369, "rewards/meter/std": 0.0778055265545845, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7613427639007568, "rewards/repeat_soft/std": 0.031513430178165436, "rewards/judge_quality/mean": 0.2887499928474426, "rewards/judge_quality/std": 0.11630470305681229, "rewards/total_composite/mean": 0.45387911796569824, "rewards/total_composite/std": 0.07856108993291855, "reward": 0.45387911796569824, "reward_std": 0.07856109738349915, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07520117610692978, "sampling/sampling_logp_difference/max": 1.9786715507507324, "sampling/importance_sampling_ratio/min": 0.13825276494026184, "sampling/importance_sampling_ratio/mean": 1.0020201206207275, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.49315447732806206, "clip_ratio/low_mean": 0.045178765431046486, "clip_ratio/low_min": 0.045178765431046486, "clip_ratio/high_mean": 0.024684758856892586, "clip_ratio/high_max": 0.024684758856892586, "clip_ratio/region_mean": 0.06986352428793907, "reward_total_mean": 0.45387911796569824, "reward_meter_mean": 0.9501059055328369, "reward_meter_std": 0.0778055265545845, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7613427639007568, "reward_repeat_soft_std": 0.031513430178165436, "reward_judge_quality_mean": 0.2887499928474426, "reward_judge_quality_std": 0.11630470305681229, "reward_total_composite_mean": 0.45387911796569824, "reward_total_composite_std": 0.07856108993291855} {"timestamp_utc": "2026-04-13T11:44:23Z", "mode": "train", "global_step": 1786, "epoch": 0.1794073329984932, "loss": 0.2066, "grad_norm": 12.4069242477417, "learning_rate": 4.590909090909092e-06, "num_tokens": 3189002.0, "completions/mean_length": 58.0, "completions/min_length": 44.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.8881990909576416, "rewards/meter/std": 0.18764106929302216, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9975974559783936, "rewards/repeat_soft/std": 0.002893527038395405, "rewards/judge_quality/mean": 0.6775000095367432, "rewards/judge_quality/std": 0.2271406203508377, "rewards/total_composite/mean": 0.7222679853439331, "rewards/total_composite/std": 0.16378124058246613, "reward": 0.7222679853439331, "reward_std": 0.16378121078014374, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15097936987876892, "sampling/sampling_logp_difference/max": 1.3733134269714355, "sampling/importance_sampling_ratio/min": 0.2532663643360138, "sampling/importance_sampling_ratio/mean": 0.993457019329071, "sampling/importance_sampling_ratio/max": 1.794889211654663, "entropy": 1.1973204538226128, "clip_ratio/low_mean": 0.049005559645593166, "clip_ratio/low_min": 0.049005559645593166, "clip_ratio/high_mean": 0.09182675927877426, "clip_ratio/high_max": 0.09182675927877426, "clip_ratio/region_mean": 0.14083231892436743, "reward_total_mean": 0.7222679853439331, "reward_meter_mean": 0.8881990909576416, "reward_meter_std": 0.18764106929302216, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9975974559783936, "reward_repeat_soft_std": 0.002893527038395405, "reward_judge_quality_mean": 0.6775000095367432, "reward_judge_quality_std": 0.2271406203508377, "reward_total_composite_mean": 0.7222679853439331, "reward_total_composite_std": 0.16378124058246613} {"timestamp_utc": "2026-04-13T11:44:31Z", "mode": "train", "global_step": 1787, "epoch": 0.17950778503264692, "loss": 0.0232, "grad_norm": 5.080633163452148, "learning_rate": 4.587878787878788e-06, "num_tokens": 3191384.0, "completions/mean_length": 117.75, "completions/min_length": 113.0, "completions/max_length": 131.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 117.75, "completions/min_terminated_length": 113.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.8378202319145203, "rewards/meter/std": 0.16670092940330505, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6724264621734619, "rewards/repeat_soft/std": 0.07522182166576385, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.4547361731529236, "rewards/total_composite/std": 0.06651253998279572, "reward": 0.4547361731529236, "reward_std": 0.06651252508163452, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.061867937445640564, "sampling/sampling_logp_difference/max": 2.2654361724853516, "sampling/importance_sampling_ratio/min": 0.10378475487232208, "sampling/importance_sampling_ratio/mean": 0.9977706670761108, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.29621799662709236, "clip_ratio/low_mean": 0.0259244404733181, "clip_ratio/low_min": 0.0259244404733181, "clip_ratio/high_mean": 0.02461230894550681, "clip_ratio/high_max": 0.02461230894550681, "clip_ratio/region_mean": 0.05053674941882491, "reward_total_mean": 0.4547361731529236, "reward_meter_mean": 0.8378202319145203, "reward_meter_std": 0.16670092940330505, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6724264621734619, "reward_repeat_soft_std": 0.07522182166576385, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.4547361731529236, "reward_total_composite_std": 0.06651253998279572} {"timestamp_utc": "2026-04-13T11:44:40Z", "mode": "train", "global_step": 1788, "epoch": 0.1796082370668006, "loss": 0.0396, "grad_norm": 5.669865608215332, "learning_rate": 4.5848484848484854e-06, "num_tokens": 3194393.0, "completions/mean_length": 180.125, "completions/min_length": 149.0, "completions/max_length": 213.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 180.125, "completions/min_terminated_length": 149.0, "completions/max_terminated_length": 213.0, "rewards/meter/mean": 0.9179707765579224, "rewards/meter/std": 0.1364118903875351, "rewards/count_adherence/mean": 0.5625, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7013819813728333, "rewards/repeat_soft/std": 0.09097636491060257, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.10966669768095016, "rewards/total_composite/mean": 0.42079004645347595, "rewards/total_composite/std": 0.07772660255432129, "reward": 0.42079004645347595, "reward_std": 0.07772661000490189, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0762377679347992, "sampling/sampling_logp_difference/max": 2.0151617527008057, "sampling/importance_sampling_ratio/min": 0.1332988440990448, "sampling/importance_sampling_ratio/mean": 1.004361629486084, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.45144612342119217, "clip_ratio/low_mean": 0.0337425353936851, "clip_ratio/low_min": 0.0337425353936851, "clip_ratio/high_mean": 0.037409089505672455, "clip_ratio/high_max": 0.037409089505672455, "clip_ratio/region_mean": 0.07115162489935756, "reward_total_mean": 0.42079004645347595, "reward_meter_mean": 0.9179707765579224, "reward_meter_std": 0.1364118903875351, "reward_count_adherence_mean": 0.5625, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7013819813728333, "reward_repeat_soft_std": 0.09097636491060257, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.10966669768095016, "reward_total_composite_mean": 0.42079004645347595, "reward_total_composite_std": 0.07772660255432129} {"timestamp_utc": "2026-04-13T11:44:47Z", "mode": "train", "global_step": 1789, "epoch": 0.1797086891009543, "loss": -0.0047, "grad_norm": 4.517831325531006, "learning_rate": 4.581818181818183e-06, "num_tokens": 3196695.0, "completions/mean_length": 112.75, "completions/min_length": 105.0, "completions/max_length": 117.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.75, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.7732725143432617, "rewards/meter/std": 0.24383077025413513, "rewards/count_adherence/mean": 0.53125, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.628362238407135, "rewards/repeat_soft/std": 0.0992504432797432, "rewards/judge_quality/mean": 0.35249999165534973, "rewards/judge_quality/std": 0.1249857097864151, "rewards/total_composite/mean": 0.3709080219268799, "rewards/total_composite/std": 0.0777900367975235, "reward": 0.3709080219268799, "reward_std": 0.0777900367975235, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06778769195079803, "sampling/sampling_logp_difference/max": 2.028944492340088, "sampling/importance_sampling_ratio/min": 0.13147422671318054, "sampling/importance_sampling_ratio/mean": 0.9983298182487488, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.34576996602118015, "clip_ratio/low_mean": 0.028252380900084972, "clip_ratio/low_min": 0.028252380900084972, "clip_ratio/high_mean": 0.041362200397998095, "clip_ratio/high_max": 0.041362200397998095, "clip_ratio/region_mean": 0.06961458129808307, "reward_total_mean": 0.3709080219268799, "reward_meter_mean": 0.7732725143432617, "reward_meter_std": 0.24383077025413513, "reward_count_adherence_mean": 0.53125, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.628362238407135, "reward_repeat_soft_std": 0.0992504432797432, "reward_judge_quality_mean": 0.35249999165534973, "reward_judge_quality_std": 0.1249857097864151, "reward_total_composite_mean": 0.3709080219268799, "reward_total_composite_std": 0.0777900367975235} {"timestamp_utc": "2026-04-13T11:44:53Z", "mode": "train", "global_step": 1790, "epoch": 0.17980914113510799, "loss": 0.0722, "grad_norm": 11.943293571472168, "learning_rate": 4.578787878787879e-06, "num_tokens": 3198182.0, "completions/mean_length": 26.875, "completions/min_length": 24.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.875, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.590266227722168, "rewards/meter/std": 0.448323518037796, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9335227012634277, "rewards/repeat_soft/std": 0.04519963264465332, "rewards/judge_quality/mean": 0.6225000023841858, "rewards/judge_quality/std": 0.24656209349632263, "rewards/total_composite/mean": 0.5792888402938843, "rewards/total_composite/std": 0.19423989951610565, "reward": 0.5792888402938843, "reward_std": 0.19423988461494446, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13793045282363892, "sampling/sampling_logp_difference/max": 1.7279443740844727, "sampling/importance_sampling_ratio/min": 0.17764921486377716, "sampling/importance_sampling_ratio/mean": 1.0088661909103394, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9380409270524979, "clip_ratio/low_mean": 0.04325311351567507, "clip_ratio/low_min": 0.04325311351567507, "clip_ratio/high_mean": 0.07876068446785212, "clip_ratio/high_max": 0.07876068446785212, "clip_ratio/region_mean": 0.12201379798352718, "reward_total_mean": 0.5792888402938843, "reward_meter_mean": 0.590266227722168, "reward_meter_std": 0.448323518037796, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9335227012634277, "reward_repeat_soft_std": 0.04519963264465332, "reward_judge_quality_mean": 0.6225000023841858, "reward_judge_quality_std": 0.24656209349632263, "reward_total_composite_mean": 0.5792888402938843, "reward_total_composite_std": 0.19423989951610565} {"timestamp_utc": "2026-04-13T11:45:02Z", "mode": "train", "global_step": 1791, "epoch": 0.17990959316926167, "loss": -0.0217, "grad_norm": 4.378878593444824, "learning_rate": 4.575757575757576e-06, "num_tokens": 3201323.0, "completions/mean_length": 183.625, "completions/min_length": 142.0, "completions/max_length": 223.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 183.625, "completions/min_terminated_length": 142.0, "completions/max_terminated_length": 223.0, "rewards/meter/mean": 0.9574471712112427, "rewards/meter/std": 0.07274672389030457, "rewards/count_adherence/mean": 0.6000000238418579, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7691037654876709, "rewards/repeat_soft/std": 0.11240530014038086, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.469510018825531, "rewards/total_composite/std": 0.05962883681058884, "reward": 0.469510018825531, "reward_std": 0.05962882936000824, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08641237020492554, "sampling/sampling_logp_difference/max": 1.7538394927978516, "sampling/importance_sampling_ratio/min": 0.25902318954467773, "sampling/importance_sampling_ratio/mean": 1.0056226253509521, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5196529664099216, "clip_ratio/low_mean": 0.024049400817602873, "clip_ratio/low_min": 0.024049400817602873, "clip_ratio/high_mean": 0.06004335265606642, "clip_ratio/high_max": 0.06004335265606642, "clip_ratio/region_mean": 0.08409275347366929, "reward_total_mean": 0.469510018825531, "reward_meter_mean": 0.9574471712112427, "reward_meter_std": 0.07274672389030457, "reward_count_adherence_mean": 0.6000000238418579, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7691037654876709, "reward_repeat_soft_std": 0.11240530014038086, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.469510018825531, "reward_total_composite_std": 0.05962883681058884} {"timestamp_utc": "2026-04-13T11:45:09Z", "mode": "train", "global_step": 1792, "epoch": 0.18001004520341538, "loss": -0.09, "grad_norm": 8.902054786682129, "learning_rate": 4.572727272727273e-06, "num_tokens": 3203121.0, "completions/mean_length": 62.75, "completions/min_length": 42.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.75, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.7635539174079895, "rewards/meter/std": 0.3895902633666992, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8927844166755676, "rewards/repeat_soft/std": 0.032342758029699326, "rewards/judge_quality/mean": 0.5275000333786011, "rewards/judge_quality/std": 0.2499571591615677, "rewards/total_composite/mean": 0.6104133129119873, "rewards/total_composite/std": 0.21616396307945251, "reward": 0.6104133129119873, "reward_std": 0.21616393327713013, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11459516733884811, "sampling/sampling_logp_difference/max": 1.3093681335449219, "sampling/importance_sampling_ratio/min": 0.26999059319496155, "sampling/importance_sampling_ratio/mean": 1.02434241771698, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0145499035716057, "clip_ratio/low_mean": 0.0704440912231803, "clip_ratio/low_min": 0.0704440912231803, "clip_ratio/high_mean": 0.025892728939652443, "clip_ratio/high_max": 0.025892728939652443, "clip_ratio/region_mean": 0.09633682016283274, "reward_total_mean": 0.6104133129119873, "reward_meter_mean": 0.7635539174079895, "reward_meter_std": 0.3895902633666992, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8927844166755676, "reward_repeat_soft_std": 0.032342758029699326, "reward_judge_quality_mean": 0.5275000333786011, "reward_judge_quality_std": 0.2499571591615677, "reward_total_composite_mean": 0.6104133129119873, "reward_total_composite_std": 0.21616396307945251} {"timestamp_utc": "2026-04-13T11:45:16Z", "mode": "train", "global_step": 1793, "epoch": 0.18011049723756906, "loss": 0.0053, "grad_norm": 12.37386703491211, "learning_rate": 4.56969696969697e-06, "num_tokens": 3204599.0, "completions/mean_length": 34.75, "completions/min_length": 32.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.75, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.964697539806366, "rewards/meter/std": 0.003814697964116931, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8172740936279297, "rewards/repeat_soft/std": 0.10937722027301788, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.6250307559967041, "rewards/total_composite/std": 0.12092455476522446, "reward": 0.6250307559967041, "reward_std": 0.12092454731464386, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06922697275876999, "sampling/sampling_logp_difference/max": 1.4014520645141602, "sampling/importance_sampling_ratio/min": 0.24623914062976837, "sampling/importance_sampling_ratio/mean": 1.0026189088821411, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.35333433747291565, "clip_ratio/low_mean": 0.06949255149811506, "clip_ratio/low_min": 0.06949255149811506, "clip_ratio/high_mean": 0.013888888992369175, "clip_ratio/high_max": 0.013888888992369175, "clip_ratio/region_mean": 0.08338144049048424, "reward_total_mean": 0.6250307559967041, "reward_meter_mean": 0.964697539806366, "reward_meter_std": 0.003814697964116931, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8172740936279297, "reward_repeat_soft_std": 0.10937722027301788, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.6250307559967041, "reward_total_composite_std": 0.12092455476522446} {"timestamp_utc": "2026-04-13T11:45:23Z", "mode": "train", "global_step": 1794, "epoch": 0.18021094927172276, "loss": 0.042, "grad_norm": 8.380064010620117, "learning_rate": 4.566666666666667e-06, "num_tokens": 3206257.0, "completions/mean_length": 47.25, "completions/min_length": 40.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.25, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9883179068565369, "rewards/meter/std": 0.0074955374002456665, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8761883974075317, "rewards/repeat_soft/std": 0.03360023722052574, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.6012390851974487, "rewards/total_composite/std": 0.0041947257705032825, "reward": 0.6012390851974487, "reward_std": 0.004194718785583973, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10523198544979095, "sampling/sampling_logp_difference/max": 1.8004121780395508, "sampling/importance_sampling_ratio/min": 0.16523078083992004, "sampling/importance_sampling_ratio/mean": 1.0199353694915771, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7152277380228043, "clip_ratio/low_mean": 0.04757572757080197, "clip_ratio/low_min": 0.04757572757080197, "clip_ratio/high_mean": 0.040652749594300985, "clip_ratio/high_max": 0.040652749594300985, "clip_ratio/region_mean": 0.08822847716510296, "reward_total_mean": 0.6012390851974487, "reward_meter_mean": 0.9883179068565369, "reward_meter_std": 0.0074955374002456665, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8761883974075317, "reward_repeat_soft_std": 0.03360023722052574, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.6012390851974487, "reward_total_composite_std": 0.0041947257705032825} {"timestamp_utc": "2026-04-13T11:45:31Z", "mode": "train", "global_step": 1795, "epoch": 0.18031140130587645, "loss": 0.0115, "grad_norm": 6.115444660186768, "learning_rate": 4.563636363636364e-06, "num_tokens": 3208312.0, "completions/mean_length": 101.875, "completions/min_length": 90.0, "completions/max_length": 113.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 101.875, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 113.0, "rewards/meter/mean": 0.8848729133605957, "rewards/meter/std": 0.18064017593860626, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8048514127731323, "rewards/repeat_soft/std": 0.07562015950679779, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.5051406621932983, "rewards/total_composite/std": 0.09334918111562729, "reward": 0.5051406621932983, "reward_std": 0.09334918111562729, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10673120617866516, "sampling/sampling_logp_difference/max": 1.910182237625122, "sampling/importance_sampling_ratio/min": 0.14805340766906738, "sampling/importance_sampling_ratio/mean": 1.0092767477035522, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6201447658240795, "clip_ratio/low_mean": 0.04387442488223314, "clip_ratio/low_min": 0.04387442488223314, "clip_ratio/high_mean": 0.045673233456909657, "clip_ratio/high_max": 0.045673233456909657, "clip_ratio/region_mean": 0.0895476583391428, "reward_total_mean": 0.5051406621932983, "reward_meter_mean": 0.8848729133605957, "reward_meter_std": 0.18064017593860626, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8048514127731323, "reward_repeat_soft_std": 0.07562015950679779, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.5051406621932983, "reward_total_composite_std": 0.09334918111562729} {"timestamp_utc": "2026-04-13T11:45:39Z", "mode": "train", "global_step": 1796, "epoch": 0.18041185334003013, "loss": 0.065, "grad_norm": 5.339907169342041, "learning_rate": 4.560606060606061e-06, "num_tokens": 3210589.0, "completions/mean_length": 119.625, "completions/min_length": 104.0, "completions/max_length": 157.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 119.625, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 157.0, "rewards/meter/mean": 0.9552260637283325, "rewards/meter/std": 0.07213979214429855, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.1178511381149292, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8133866786956787, "rewards/repeat_soft/std": 0.061037372797727585, "rewards/judge_quality/mean": 0.44874998927116394, "rewards/judge_quality/std": 0.16137246787548065, "rewards/total_composite/mean": 0.5194798707962036, "rewards/total_composite/std": 0.07868607342243195, "reward": 0.5194798707962036, "reward_std": 0.07868607342243195, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08553514629602432, "sampling/sampling_logp_difference/max": 3.378880500793457, "sampling/importance_sampling_ratio/min": 0.03408559411764145, "sampling/importance_sampling_ratio/mean": 1.0060369968414307, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5528635047376156, "clip_ratio/low_mean": 0.04500382672995329, "clip_ratio/low_min": 0.04500382672995329, "clip_ratio/high_mean": 0.03737408481538296, "clip_ratio/high_max": 0.03737408481538296, "clip_ratio/region_mean": 0.08237791154533625, "reward_total_mean": 0.5194798707962036, "reward_meter_mean": 0.9552260637283325, "reward_meter_std": 0.07213979214429855, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.1178511381149292, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8133866786956787, "reward_repeat_soft_std": 0.061037372797727585, "reward_judge_quality_mean": 0.44874998927116394, "reward_judge_quality_std": 0.16137246787548065, "reward_total_composite_mean": 0.5194798707962036, "reward_total_composite_std": 0.07868607342243195} {"timestamp_utc": "2026-04-13T11:45:50Z", "mode": "train", "global_step": 1797, "epoch": 0.18051230537418383, "loss": -0.2245, "grad_norm": 2.382840156555176, "learning_rate": 4.557575757575758e-06, "num_tokens": 3213457.0, "completions/mean_length": 231.5, "completions/min_length": 164.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 191.42857360839844, "completions/min_terminated_length": 164.0, "completions/max_terminated_length": 223.0, "rewards/meter/mean": 0.6678950190544128, "rewards/meter/std": 0.4299687445163727, "rewards/count_adherence/mean": 0.7250000238418579, "rewards/count_adherence/std": 0.1035098284482956, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8232831358909607, "rewards/repeat_soft/std": 0.07086049765348434, "rewards/judge_quality/mean": 0.3399999737739563, "rewards/judge_quality/std": 0.1505228877067566, "rewards/total_composite/mean": 0.3902186155319214, "rewards/total_composite/std": 0.18478281795978546, "reward": 0.3902186155319214, "reward_std": 0.18478280305862427, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09489186853170395, "sampling/sampling_logp_difference/max": 1.3858833312988281, "sampling/importance_sampling_ratio/min": 0.25010278820991516, "sampling/importance_sampling_ratio/mean": 1.0041186809539795, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5956591069698334, "clip_ratio/low_mean": 0.021478006150573492, "clip_ratio/low_min": 0.021478006150573492, "clip_ratio/high_mean": 0.05395158566534519, "clip_ratio/high_max": 0.05395158566534519, "clip_ratio/region_mean": 0.07542959181591868, "reward_total_mean": 0.3902186155319214, "reward_meter_mean": 0.6678950190544128, "reward_meter_std": 0.4299687445163727, "reward_count_adherence_mean": 0.7250000238418579, "reward_count_adherence_std": 0.1035098284482956, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8232831358909607, "reward_repeat_soft_std": 0.07086049765348434, "reward_judge_quality_mean": 0.3399999737739563, "reward_judge_quality_std": 0.1505228877067566, "reward_total_composite_mean": 0.3902186155319214, "reward_total_composite_std": 0.18478281795978546} {"timestamp_utc": "2026-04-13T11:45:58Z", "mode": "train", "global_step": 1798, "epoch": 0.18061275740833752, "loss": 0.2943, "grad_norm": 11.488846778869629, "learning_rate": 4.554545454545455e-06, "num_tokens": 3215041.0, "completions/mean_length": 48.0, "completions/min_length": 27.0, "completions/max_length": 92.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.0, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 92.0, "rewards/meter/mean": 0.857647716999054, "rewards/meter/std": 0.15262115001678467, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.37796446681022644, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9310461282730103, "rewards/repeat_soft/std": 0.08830144256353378, "rewards/judge_quality/mean": 0.4937500059604645, "rewards/judge_quality/std": 0.1728696972131729, "rewards/total_composite/mean": 0.5560691952705383, "rewards/total_composite/std": 0.11330970376729965, "reward": 0.5560691952705383, "reward_std": 0.11330970376729965, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13383953273296356, "sampling/sampling_logp_difference/max": 1.622748851776123, "sampling/importance_sampling_ratio/min": 0.1973554491996765, "sampling/importance_sampling_ratio/mean": 1.0150392055511475, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0053318589925766, "clip_ratio/low_mean": 0.04046074906364083, "clip_ratio/low_min": 0.04046074906364083, "clip_ratio/high_mean": 0.08967884071171284, "clip_ratio/high_max": 0.08967884071171284, "clip_ratio/region_mean": 0.13013958977535367, "reward_total_mean": 0.5560691952705383, "reward_meter_mean": 0.857647716999054, "reward_meter_std": 0.15262115001678467, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.37796446681022644, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9310461282730103, "reward_repeat_soft_std": 0.08830144256353378, "reward_judge_quality_mean": 0.4937500059604645, "reward_judge_quality_std": 0.1728696972131729, "reward_total_composite_mean": 0.5560691952705383, "reward_total_composite_std": 0.11330970376729965} {"timestamp_utc": "2026-04-13T11:46:06Z", "mode": "train", "global_step": 1799, "epoch": 0.18071320944249122, "loss": 0.0474, "grad_norm": 7.578016757965088, "learning_rate": 4.551515151515152e-06, "num_tokens": 3217700.0, "completions/mean_length": 167.375, "completions/min_length": 136.0, "completions/max_length": 192.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 167.375, "completions/min_terminated_length": 136.0, "completions/max_terminated_length": 192.0, "rewards/meter/mean": 0.9869016408920288, "rewards/meter/std": 0.010118884965777397, "rewards/count_adherence/mean": 0.53125, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6848177909851074, "rewards/repeat_soft/std": 0.0947936475276947, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.4647965431213379, "rewards/total_composite/std": 0.040095239877700806, "reward": 0.4647965431213379, "reward_std": 0.040095243602991104, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07609271258115768, "sampling/sampling_logp_difference/max": 1.9342234134674072, "sampling/importance_sampling_ratio/min": 0.14453648030757904, "sampling/importance_sampling_ratio/mean": 1.0183758735656738, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5007193610072136, "clip_ratio/low_mean": 0.021510645747184753, "clip_ratio/low_min": 0.021510645747184753, "clip_ratio/high_mean": 0.04436182277277112, "clip_ratio/high_max": 0.04436182277277112, "clip_ratio/region_mean": 0.06587246851995587, "reward_total_mean": 0.4647965431213379, "reward_meter_mean": 0.9869016408920288, "reward_meter_std": 0.010118884965777397, "reward_count_adherence_mean": 0.53125, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6848177909851074, "reward_repeat_soft_std": 0.0947936475276947, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.4647965431213379, "reward_total_composite_std": 0.040095239877700806} {"timestamp_utc": "2026-04-13T11:46:14Z", "mode": "train", "global_step": 1800, "epoch": 0.1808136614766449, "loss": 0.0076, "grad_norm": 4.887317180633545, "learning_rate": 4.548484848484849e-06, "num_tokens": 3220775.0, "completions/mean_length": 200.375, "completions/min_length": 162.0, "completions/max_length": 242.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 200.375, "completions/min_terminated_length": 162.0, "completions/max_terminated_length": 242.0, "rewards/meter/mean": 0.8649101257324219, "rewards/meter/std": 0.2364814579486847, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8354235887527466, "rewards/repeat_soft/std": 0.04172105714678764, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5114339590072632, "rewards/total_composite/std": 0.06003821641206741, "reward": 0.5114339590072632, "reward_std": 0.060038212686777115, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10037865489721298, "sampling/sampling_logp_difference/max": 1.4967641830444336, "sampling/importance_sampling_ratio/min": 0.22385334968566895, "sampling/importance_sampling_ratio/mean": 1.0184109210968018, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8853504583239555, "clip_ratio/low_mean": 0.027874689549207687, "clip_ratio/low_min": 0.027874689549207687, "clip_ratio/high_mean": 0.0581750962883234, "clip_ratio/high_max": 0.0581750962883234, "clip_ratio/region_mean": 0.08604978583753109, "reward_total_mean": 0.5114339590072632, "reward_meter_mean": 0.8649101257324219, "reward_meter_std": 0.2364814579486847, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8354235887527466, "reward_repeat_soft_std": 0.04172105714678764, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5114339590072632, "reward_total_composite_std": 0.06003821641206741} {"timestamp_utc": "2026-04-13T11:47:03Z", "mode": "eval", "global_step": 1800, "epoch": 0.1808136614766449, "eval_loss": NaN, "eval_runtime": 48.3282, "eval_samples_per_second": 1.655, "eval_steps_per_second": 0.207, "eval_num_tokens": 3220775.0, "eval_completions/mean_length": 103.4125, "eval_completions/min_length": 34.7, "eval_completions/max_length": 180.3, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 103.4125, "eval_completions/min_terminated_length": 34.7, "eval_completions/max_terminated_length": 180.3, "eval_rewards/meter/mean": 0.7815804183483124, "eval_rewards/meter/std": 0.280592598952353, "eval_rewards/count_adherence/mean": 0.7295833468437195, "eval_rewards/count_adherence/std": 0.21119236797094346, "eval_rewards/hard_gate/mean": 1.0, "eval_rewards/hard_gate/std": 0.0, "eval_rewards/repeat_soft/mean": 0.8449050545692444, "eval_rewards/repeat_soft/std": 0.09524145871400833, "eval_rewards/judge_quality/mean": 0.47274999916553495, "eval_rewards/judge_quality/std": 0.15114980656653643, "eval_rewards/total_composite/mean": 0.5103550374507904, "eval_rewards/total_composite/std": 0.13814009353518486, "eval_reward": 0.5103550374507904, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.0535921260714531, "eval_sampling/sampling_logp_difference/max": 0.9655992984771729, "eval_sampling/importance_sampling_ratio/min": 0.38459205329418183, "eval_sampling/importance_sampling_ratio/mean": 1.0134925723075867, "eval_sampling/importance_sampling_ratio/max": 1.4427119255065919, "eval_entropy": 0.592803618311882, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5103550374507904, "eval_reward_meter_mean": 0.7815804183483124, "eval_reward_meter_std": 0.280592598952353, "eval_reward_count_adherence_mean": 0.7295833468437195, "eval_reward_count_adherence_std": 0.21119236797094346, "eval_reward_hard_gate_mean": 1.0, "eval_reward_hard_gate_std": 0.0, "eval_reward_repeat_soft_mean": 0.8449050545692444, "eval_reward_repeat_soft_std": 0.09524145871400833, "eval_reward_judge_quality_mean": 0.47274999916553495, "eval_reward_judge_quality_std": 0.15114980656653643, "eval_reward_total_composite_mean": 0.5103550374507904, "eval_reward_total_composite_std": 0.13814009353518486} {"timestamp_utc": "2026-04-13T11:47:12Z", "mode": "train", "global_step": 1801, "epoch": 0.18091411351079859, "loss": 0.0177, "grad_norm": 9.770906448364258, "learning_rate": 4.5454545454545455e-06, "num_tokens": 3222243.0, "completions/mean_length": 35.5, "completions/min_length": 34.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.5, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9592167735099792, "rewards/meter/std": 0.02322445809841156, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7318859696388245, "rewards/repeat_soft/std": 0.0112870829179883, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5716490745544434, "rewards/total_composite/std": 0.0063031138852238655, "reward": 0.5716490745544434, "reward_std": 0.006303121335804462, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04629416763782501, "sampling/sampling_logp_difference/max": 1.4752837419509888, "sampling/importance_sampling_ratio/min": 0.22871382534503937, "sampling/importance_sampling_ratio/mean": 1.0094786882400513, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2513156123459339, "clip_ratio/low_mean": 0.010135134682059288, "clip_ratio/low_min": 0.010135134682059288, "clip_ratio/high_mean": 0.02460317499935627, "clip_ratio/high_max": 0.02460317499935627, "clip_ratio/region_mean": 0.03473830968141556, "reward_total_mean": 0.5716490745544434, "reward_meter_mean": 0.9592167735099792, "reward_meter_std": 0.02322445809841156, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7318859696388245, "reward_repeat_soft_std": 0.0112870829179883, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5716490745544434, "reward_total_composite_std": 0.0063031138852238655} {"timestamp_utc": "2026-04-13T11:47:20Z", "mode": "train", "global_step": 1802, "epoch": 0.1810145655449523, "loss": 0.0007, "grad_norm": 5.378952503204346, "learning_rate": 4.542424242424243e-06, "num_tokens": 3225235.0, "completions/mean_length": 184.0, "completions/min_length": 156.0, "completions/max_length": 209.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 184.0, "completions/min_terminated_length": 156.0, "completions/max_terminated_length": 209.0, "rewards/meter/mean": 0.7820342183113098, "rewards/meter/std": 0.33812469244003296, "rewards/count_adherence/mean": 0.5625, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8114336729049683, "rewards/repeat_soft/std": 0.049185801297426224, "rewards/judge_quality/mean": 0.3687500059604645, "rewards/judge_quality/std": 0.10802611708641052, "rewards/total_composite/mean": 0.4131154417991638, "rewards/total_composite/std": 0.09631349891424179, "reward": 0.4131154417991638, "reward_std": 0.09631349891424179, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10045631974935532, "sampling/sampling_logp_difference/max": 1.3578910827636719, "sampling/importance_sampling_ratio/min": 0.2572026252746582, "sampling/importance_sampling_ratio/mean": 1.0220563411712646, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9590811803936958, "clip_ratio/low_mean": 0.0426053018309176, "clip_ratio/low_min": 0.0426053018309176, "clip_ratio/high_mean": 0.044604139402508736, "clip_ratio/high_max": 0.044604139402508736, "clip_ratio/region_mean": 0.08720944123342633, "reward_total_mean": 0.4131154417991638, "reward_meter_mean": 0.7820342183113098, "reward_meter_std": 0.33812469244003296, "reward_count_adherence_mean": 0.5625, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8114336729049683, "reward_repeat_soft_std": 0.049185801297426224, "reward_judge_quality_mean": 0.3687500059604645, "reward_judge_quality_std": 0.10802611708641052, "reward_total_composite_mean": 0.4131154417991638, "reward_total_composite_std": 0.09631349891424179} {"timestamp_utc": "2026-04-13T11:47:28Z", "mode": "train", "global_step": 1803, "epoch": 0.18111501757910597, "loss": 0.0786, "grad_norm": 11.218101501464844, "learning_rate": 4.539393939393939e-06, "num_tokens": 3227036.0, "completions/mean_length": 65.125, "completions/min_length": 52.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.125, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.6598147749900818, "rewards/meter/std": 0.3006100654602051, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8704584836959839, "rewards/repeat_soft/std": 0.06427149474620819, "rewards/judge_quality/mean": 0.7950000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.5993191599845886, "rewards/total_composite/std": 0.1949187070131302, "reward": 0.5993191599845886, "reward_std": 0.1949187070131302, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10370650142431259, "sampling/sampling_logp_difference/max": 1.9163055419921875, "sampling/importance_sampling_ratio/min": 0.14714960753917694, "sampling/importance_sampling_ratio/mean": 0.9993343353271484, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4846961982548237, "clip_ratio/low_mean": 0.06705652084201574, "clip_ratio/low_min": 0.06705652084201574, "clip_ratio/high_mean": 0.0369061091914773, "clip_ratio/high_max": 0.0369061091914773, "clip_ratio/region_mean": 0.10396263003349304, "reward_total_mean": 0.5993191599845886, "reward_meter_mean": 0.6598147749900818, "reward_meter_std": 0.3006100654602051, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8704584836959839, "reward_repeat_soft_std": 0.06427149474620819, "reward_judge_quality_mean": 0.7950000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.5993191599845886, "reward_total_composite_std": 0.1949187070131302} {"timestamp_utc": "2026-04-13T11:47:36Z", "mode": "train", "global_step": 1804, "epoch": 0.18121546961325966, "loss": 0.0992, "grad_norm": 10.028509140014648, "learning_rate": 4.5363636363636364e-06, "num_tokens": 3228749.0, "completions/mean_length": 56.125, "completions/min_length": 44.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.125, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9606437683105469, "rewards/meter/std": 0.0385453999042511, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9285358190536499, "rewards/repeat_soft/std": 0.04393455758690834, "rewards/judge_quality/mean": 0.668749988079071, "rewards/judge_quality/std": 0.25842589139938354, "rewards/total_composite/mean": 0.7556654810905457, "rewards/total_composite/std": 0.16067703068256378, "reward": 0.7556654810905457, "reward_std": 0.16067704558372498, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11988765001296997, "sampling/sampling_logp_difference/max": 1.3343610763549805, "sampling/importance_sampling_ratio/min": 0.26332637667655945, "sampling/importance_sampling_ratio/mean": 0.998325526714325, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.775499127805233, "clip_ratio/low_mean": 0.03212598082609475, "clip_ratio/low_min": 0.03212598082609475, "clip_ratio/high_mean": 0.06991446064785123, "clip_ratio/high_max": 0.06991446064785123, "clip_ratio/region_mean": 0.10204044147394598, "reward_total_mean": 0.7556654810905457, "reward_meter_mean": 0.9606437683105469, "reward_meter_std": 0.0385453999042511, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9285358190536499, "reward_repeat_soft_std": 0.04393455758690834, "reward_judge_quality_mean": 0.668749988079071, "reward_judge_quality_std": 0.25842589139938354, "reward_total_composite_mean": 0.7556654810905457, "reward_total_composite_std": 0.16067703068256378} {"timestamp_utc": "2026-04-13T11:47:43Z", "mode": "train", "global_step": 1805, "epoch": 0.18131592164741336, "loss": -0.0555, "grad_norm": 10.76839542388916, "learning_rate": 4.533333333333334e-06, "num_tokens": 3230528.0, "completions/mean_length": 57.375, "completions/min_length": 43.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.375, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.6418136954307556, "rewards/meter/std": 0.36736324429512024, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.25877460837364197, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9308949708938599, "rewards/repeat_soft/std": 0.11350374668836594, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.5372963547706604, "rewards/total_composite/std": 0.14024575054645538, "reward": 0.5372963547706604, "reward_std": 0.14024576544761658, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1133660301566124, "sampling/sampling_logp_difference/max": 1.6175422668457031, "sampling/importance_sampling_ratio/min": 0.19838567078113556, "sampling/importance_sampling_ratio/mean": 1.0129854679107666, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8203057646751404, "clip_ratio/low_mean": 0.06033054692670703, "clip_ratio/low_min": 0.06033054692670703, "clip_ratio/high_mean": 0.04288742411881685, "clip_ratio/high_max": 0.04288742411881685, "clip_ratio/region_mean": 0.10321797104552388, "reward_total_mean": 0.5372963547706604, "reward_meter_mean": 0.6418136954307556, "reward_meter_std": 0.36736324429512024, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.25877460837364197, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9308949708938599, "reward_repeat_soft_std": 0.11350374668836594, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.5372963547706604, "reward_total_composite_std": 0.14024575054645538} {"timestamp_utc": "2026-04-13T11:47:51Z", "mode": "train", "global_step": 1806, "epoch": 0.18141637368156704, "loss": 0.0055, "grad_norm": 4.9276909828186035, "learning_rate": 4.53030303030303e-06, "num_tokens": 3233534.0, "completions/mean_length": 181.75, "completions/min_length": 146.0, "completions/max_length": 223.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 181.75, "completions/min_terminated_length": 146.0, "completions/max_terminated_length": 223.0, "rewards/meter/mean": 0.7319014072418213, "rewards/meter/std": 0.27093157172203064, "rewards/count_adherence/mean": 0.7083333730697632, "rewards/count_adherence/std": 0.07715165615081787, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7889889478683472, "rewards/repeat_soft/std": 0.03599509969353676, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4598240852355957, "rewards/total_composite/std": 0.07499600946903229, "reward": 0.4598240852355957, "reward_std": 0.07499600946903229, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08227083832025528, "sampling/sampling_logp_difference/max": 1.5979204177856445, "sampling/importance_sampling_ratio/min": 0.20231682062149048, "sampling/importance_sampling_ratio/mean": 1.0073206424713135, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6394183374941349, "clip_ratio/low_mean": 0.026449366007000208, "clip_ratio/low_min": 0.026449366007000208, "clip_ratio/high_mean": 0.052284734789282084, "clip_ratio/high_max": 0.052284734789282084, "clip_ratio/region_mean": 0.07873410079628229, "reward_total_mean": 0.4598240852355957, "reward_meter_mean": 0.7319014072418213, "reward_meter_std": 0.27093157172203064, "reward_count_adherence_mean": 0.7083333730697632, "reward_count_adherence_std": 0.07715165615081787, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7889889478683472, "reward_repeat_soft_std": 0.03599509969353676, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4598240852355957, "reward_total_composite_std": 0.07499600946903229} {"timestamp_utc": "2026-04-13T11:47:57Z", "mode": "train", "global_step": 1807, "epoch": 0.18151682571572075, "loss": 0.0895, "grad_norm": 16.027671813964844, "learning_rate": 4.527272727272727e-06, "num_tokens": 3234932.0, "completions/mean_length": 24.75, "completions/min_length": 18.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.75, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9225627779960632, "rewards/meter/std": 0.09814183413982391, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9316083192825317, "rewards/repeat_soft/std": 0.05721777305006981, "rewards/judge_quality/mean": 0.4099999964237213, "rewards/judge_quality/std": 0.06633248925209045, "rewards/total_composite/mean": 0.5849047303199768, "rewards/total_composite/std": 0.04630429670214653, "reward": 0.5849047303199768, "reward_std": 0.04630429670214653, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1425008922815323, "sampling/sampling_logp_difference/max": 1.8376264572143555, "sampling/importance_sampling_ratio/min": 0.15919482707977295, "sampling/importance_sampling_ratio/mean": 1.0083662271499634, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0856371745467186, "clip_ratio/low_mean": 0.06887955591082573, "clip_ratio/low_min": 0.06887955591082573, "clip_ratio/high_mean": 0.12037668284028769, "clip_ratio/high_max": 0.12037668284028769, "clip_ratio/region_mean": 0.18925623875111341, "reward_total_mean": 0.5849047303199768, "reward_meter_mean": 0.9225627779960632, "reward_meter_std": 0.09814183413982391, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9316083192825317, "reward_repeat_soft_std": 0.05721777305006981, "reward_judge_quality_mean": 0.4099999964237213, "reward_judge_quality_std": 0.06633248925209045, "reward_total_composite_mean": 0.5849047303199768, "reward_total_composite_std": 0.04630429670214653} {"timestamp_utc": "2026-04-13T11:48:04Z", "mode": "train", "global_step": 1808, "epoch": 0.18161727774987443, "loss": 0.0453, "grad_norm": 8.079506874084473, "learning_rate": 4.524242424242425e-06, "num_tokens": 3237030.0, "completions/mean_length": 95.25, "completions/min_length": 83.0, "completions/max_length": 111.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.25, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.9181572794914246, "rewards/meter/std": 0.12369729578495026, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8471939563751221, "rewards/repeat_soft/std": 0.021875236183404922, "rewards/judge_quality/mean": 0.8362500667572021, "rewards/judge_quality/std": 0.23688077926635742, "rewards/total_composite/mean": 0.7605818510055542, "rewards/total_composite/std": 0.15978670120239258, "reward": 0.7605818510055542, "reward_std": 0.15978671610355377, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09632844477891922, "sampling/sampling_logp_difference/max": 1.2662220001220703, "sampling/importance_sampling_ratio/min": 0.28189462423324585, "sampling/importance_sampling_ratio/mean": 1.019246220588684, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7270065546035767, "clip_ratio/low_mean": 0.019042464904487133, "clip_ratio/low_min": 0.019042464904487133, "clip_ratio/high_mean": 0.06389599200338125, "clip_ratio/high_max": 0.06389599200338125, "clip_ratio/region_mean": 0.08293845690786839, "reward_total_mean": 0.7605818510055542, "reward_meter_mean": 0.9181572794914246, "reward_meter_std": 0.12369729578495026, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8471939563751221, "reward_repeat_soft_std": 0.021875236183404922, "reward_judge_quality_mean": 0.8362500667572021, "reward_judge_quality_std": 0.23688077926635742, "reward_total_composite_mean": 0.7605818510055542, "reward_total_composite_std": 0.15978670120239258} {"timestamp_utc": "2026-04-13T11:48:12Z", "mode": "train", "global_step": 1809, "epoch": 0.18171772978402811, "loss": -0.0039, "grad_norm": 11.031441688537598, "learning_rate": 4.521212121212122e-06, "num_tokens": 3239058.0, "completions/mean_length": 90.5, "completions/min_length": 76.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.5, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.9052593111991882, "rewards/meter/std": 0.10923722386360168, "rewards/count_adherence/mean": 0.4166666865348816, "rewards/count_adherence/std": 0.15430335700511932, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8219844102859497, "rewards/repeat_soft/std": 0.07050145417451859, "rewards/judge_quality/mean": 0.41499999165534973, "rewards/judge_quality/std": 0.14520922303199768, "rewards/total_composite/mean": 0.4561029076576233, "rewards/total_composite/std": 0.10142374038696289, "reward": 0.4561029076576233, "reward_std": 0.10142374038696289, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10839082300662994, "sampling/sampling_logp_difference/max": 2.2924723625183105, "sampling/importance_sampling_ratio/min": 0.10101640224456787, "sampling/importance_sampling_ratio/mean": 1.0194932222366333, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6266350857913494, "clip_ratio/low_mean": 0.03545480594038963, "clip_ratio/low_min": 0.03545480594038963, "clip_ratio/high_mean": 0.05669600097462535, "clip_ratio/high_max": 0.05669600097462535, "clip_ratio/region_mean": 0.09215080691501498, "reward_total_mean": 0.4561029076576233, "reward_meter_mean": 0.9052593111991882, "reward_meter_std": 0.10923722386360168, "reward_count_adherence_mean": 0.4166666865348816, "reward_count_adherence_std": 0.15430335700511932, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8219844102859497, "reward_repeat_soft_std": 0.07050145417451859, "reward_judge_quality_mean": 0.41499999165534973, "reward_judge_quality_std": 0.14520922303199768, "reward_total_composite_mean": 0.4561029076576233, "reward_total_composite_std": 0.10142374038696289} {"timestamp_utc": "2026-04-13T11:48:20Z", "mode": "train", "global_step": 1810, "epoch": 0.18181818181818182, "loss": -0.025, "grad_norm": 5.897127151489258, "learning_rate": 4.518181818181819e-06, "num_tokens": 3241821.0, "completions/mean_length": 170.375, "completions/min_length": 148.0, "completions/max_length": 212.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 170.375, "completions/min_terminated_length": 148.0, "completions/max_terminated_length": 212.0, "rewards/meter/mean": 0.6426774263381958, "rewards/meter/std": 0.20387646555900574, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9584013819694519, "rewards/repeat_soft/std": 0.021950701251626015, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.4701296389102936, "rewards/total_composite/std": 0.05986081436276436, "reward": 0.4701296389102936, "reward_std": 0.05986081436276436, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13232646882534027, "sampling/sampling_logp_difference/max": 2.335085868835449, "sampling/importance_sampling_ratio/min": 0.09680216759443283, "sampling/importance_sampling_ratio/mean": 1.0158830881118774, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.075805090367794, "clip_ratio/low_mean": 0.0722449030727148, "clip_ratio/low_min": 0.0722449030727148, "clip_ratio/high_mean": 0.04664424154907465, "clip_ratio/high_max": 0.04664424154907465, "clip_ratio/region_mean": 0.11888914462178946, "reward_total_mean": 0.4701296389102936, "reward_meter_mean": 0.6426774263381958, "reward_meter_std": 0.20387646555900574, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9584013819694519, "reward_repeat_soft_std": 0.021950701251626015, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.4701296389102936, "reward_total_composite_std": 0.05986081436276436} {"timestamp_utc": "2026-04-13T11:48:27Z", "mode": "train", "global_step": 1811, "epoch": 0.1819186338523355, "loss": -0.0265, "grad_norm": 8.900130271911621, "learning_rate": 4.5151515151515155e-06, "num_tokens": 3243646.0, "completions/mean_length": 69.125, "completions/min_length": 56.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.125, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.7467904686927795, "rewards/meter/std": 0.29527267813682556, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8589749336242676, "rewards/repeat_soft/std": 0.03796645998954773, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4660533666610718, "rewards/total_composite/std": 0.0844200924038887, "reward": 0.4660533666610718, "reward_std": 0.0844200998544693, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11518388241529465, "sampling/sampling_logp_difference/max": 1.473679542541504, "sampling/importance_sampling_ratio/min": 0.3138110935688019, "sampling/importance_sampling_ratio/mean": 1.0072176456451416, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7754762470722198, "clip_ratio/low_mean": 0.04373500728979707, "clip_ratio/low_min": 0.04373500728979707, "clip_ratio/high_mean": 0.07083044201135635, "clip_ratio/high_max": 0.07083044201135635, "clip_ratio/region_mean": 0.11456544930115342, "reward_total_mean": 0.4660533666610718, "reward_meter_mean": 0.7467904686927795, "reward_meter_std": 0.29527267813682556, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8589749336242676, "reward_repeat_soft_std": 0.03796645998954773, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4660533666610718, "reward_total_composite_std": 0.0844200924038887} {"timestamp_utc": "2026-04-13T11:48:33Z", "mode": "train", "global_step": 1812, "epoch": 0.1820190858864892, "loss": 0.0516, "grad_norm": 11.70666790008545, "learning_rate": 4.512121212121213e-06, "num_tokens": 3244994.0, "completions/mean_length": 25.5, "completions/min_length": 21.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.5, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.7291864156723022, "rewards/meter/std": 0.4397353231906891, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9311109185218811, "rewards/repeat_soft/std": 0.02573113888502121, "rewards/judge_quality/mean": 0.6150000095367432, "rewards/judge_quality/std": 0.25286927819252014, "rewards/total_composite/mean": 0.62016761302948, "rewards/total_composite/std": 0.2084752470254898, "reward": 0.62016761302948, "reward_std": 0.2084752470254898, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11912678182125092, "sampling/sampling_logp_difference/max": 0.9539971351623535, "sampling/importance_sampling_ratio/min": 0.3851982653141022, "sampling/importance_sampling_ratio/mean": 1.0331926345825195, "sampling/importance_sampling_ratio/max": 1.9330251216888428, "entropy": 0.7726024240255356, "clip_ratio/low_mean": 0.04123463947325945, "clip_ratio/low_min": 0.04123463947325945, "clip_ratio/high_mean": 0.049452862702310085, "clip_ratio/high_max": 0.049452862702310085, "clip_ratio/region_mean": 0.09068750217556953, "reward_total_mean": 0.62016761302948, "reward_meter_mean": 0.7291864156723022, "reward_meter_std": 0.4397353231906891, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9311109185218811, "reward_repeat_soft_std": 0.02573113888502121, "reward_judge_quality_mean": 0.6150000095367432, "reward_judge_quality_std": 0.25286927819252014, "reward_total_composite_mean": 0.62016761302948, "reward_total_composite_std": 0.2084752470254898} {"timestamp_utc": "2026-04-13T11:48:39Z", "mode": "train", "global_step": 1813, "epoch": 0.1821195379206429, "loss": 0.0103, "grad_norm": 4.901209354400635, "learning_rate": 4.50909090909091e-06, "num_tokens": 3246798.0, "completions/mean_length": 76.5, "completions/min_length": 70.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.5, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.9378914833068848, "rewards/meter/std": 0.040167856961488724, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7148953676223755, "rewards/repeat_soft/std": 0.04254239797592163, "rewards/judge_quality/mean": 0.3687500059604645, "rewards/judge_quality/std": 0.10802611708641052, "rewards/total_composite/mean": 0.46721765398979187, "rewards/total_composite/std": 0.06920699775218964, "reward": 0.46721765398979187, "reward_std": 0.06920700520277023, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06374968588352203, "sampling/sampling_logp_difference/max": 0.9852927923202515, "sampling/importance_sampling_ratio/min": 0.37332990765571594, "sampling/importance_sampling_ratio/mean": 1.0055588483810425, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.36710847169160843, "clip_ratio/low_mean": 0.01134446682408452, "clip_ratio/low_min": 0.01134446682408452, "clip_ratio/high_mean": 0.04556079301983118, "clip_ratio/high_max": 0.04556079301983118, "clip_ratio/region_mean": 0.0569052598439157, "reward_total_mean": 0.46721765398979187, "reward_meter_mean": 0.9378914833068848, "reward_meter_std": 0.040167856961488724, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7148953676223755, "reward_repeat_soft_std": 0.04254239797592163, "reward_judge_quality_mean": 0.3687500059604645, "reward_judge_quality_std": 0.10802611708641052, "reward_total_composite_mean": 0.46721765398979187, "reward_total_composite_std": 0.06920699775218964} {"timestamp_utc": "2026-04-13T11:48:47Z", "mode": "train", "global_step": 1814, "epoch": 0.18221998995479657, "loss": 0.0555, "grad_norm": 4.9915266036987305, "learning_rate": 4.5060606060606065e-06, "num_tokens": 3249228.0, "completions/mean_length": 125.75, "completions/min_length": 107.0, "completions/max_length": 154.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 125.75, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 154.0, "rewards/meter/mean": 0.780795156955719, "rewards/meter/std": 0.2588658928871155, "rewards/count_adherence/mean": 0.5, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7797104120254517, "rewards/repeat_soft/std": 0.04299363121390343, "rewards/judge_quality/mean": 0.6450000405311584, "rewards/judge_quality/std": 0.19820626080036163, "rewards/total_composite/mean": 0.5325644612312317, "rewards/total_composite/std": 0.13640287518501282, "reward": 0.5325644612312317, "reward_std": 0.136402890086174, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07033336907625198, "sampling/sampling_logp_difference/max": 1.2939739227294922, "sampling/importance_sampling_ratio/min": 0.27417904138565063, "sampling/importance_sampling_ratio/mean": 1.0076937675476074, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5893510244786739, "clip_ratio/low_mean": 0.03880102909170091, "clip_ratio/low_min": 0.03880102909170091, "clip_ratio/high_mean": 0.01608090242370963, "clip_ratio/high_max": 0.01608090242370963, "clip_ratio/region_mean": 0.05488193151541054, "reward_total_mean": 0.5325644612312317, "reward_meter_mean": 0.780795156955719, "reward_meter_std": 0.2588658928871155, "reward_count_adherence_mean": 0.5, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7797104120254517, "reward_repeat_soft_std": 0.04299363121390343, "reward_judge_quality_mean": 0.6450000405311584, "reward_judge_quality_std": 0.19820626080036163, "reward_total_composite_mean": 0.5325644612312317, "reward_total_composite_std": 0.13640287518501282} {"timestamp_utc": "2026-04-13T11:48:54Z", "mode": "train", "global_step": 1815, "epoch": 0.18232044198895028, "loss": 0.0526, "grad_norm": 14.766071319580078, "learning_rate": 4.503030303030304e-06, "num_tokens": 3250851.0, "completions/mean_length": 47.875, "completions/min_length": 44.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.875, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.8778351545333862, "rewards/meter/std": 0.2011054903268814, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9835158586502075, "rewards/repeat_soft/std": 0.01885959878563881, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.6672040224075317, "rewards/total_composite/std": 0.17567037045955658, "reward": 0.6672040224075317, "reward_std": 0.17567035555839539, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1272457093000412, "sampling/sampling_logp_difference/max": 1.2632765769958496, "sampling/importance_sampling_ratio/min": 0.28272613883018494, "sampling/importance_sampling_ratio/mean": 1.005624771118164, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8602741062641144, "clip_ratio/low_mean": 0.06548884371295571, "clip_ratio/low_min": 0.06548884371295571, "clip_ratio/high_mean": 0.02700046170502901, "clip_ratio/high_max": 0.02700046170502901, "clip_ratio/region_mean": 0.09248930541798472, "reward_total_mean": 0.6672040224075317, "reward_meter_mean": 0.8778351545333862, "reward_meter_std": 0.2011054903268814, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9835158586502075, "reward_repeat_soft_std": 0.01885959878563881, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.6672040224075317, "reward_total_composite_std": 0.17567037045955658} {"timestamp_utc": "2026-04-13T11:49:01Z", "mode": "train", "global_step": 1816, "epoch": 0.18242089402310396, "loss": -0.0518, "grad_norm": 5.025565147399902, "learning_rate": 4.5e-06, "num_tokens": 3253580.0, "completions/mean_length": 153.125, "completions/min_length": 120.0, "completions/max_length": 188.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 153.125, "completions/min_terminated_length": 120.0, "completions/max_terminated_length": 188.0, "rewards/meter/mean": 0.8800153732299805, "rewards/meter/std": 0.16372206807136536, "rewards/count_adherence/mean": 0.5, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.788138747215271, "rewards/repeat_soft/std": 0.09071363508701324, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.45846500992774963, "rewards/total_composite/std": 0.04814787209033966, "reward": 0.45846500992774963, "reward_std": 0.04814785718917847, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09667257964611053, "sampling/sampling_logp_difference/max": 2.0956380367279053, "sampling/importance_sampling_ratio/min": 0.12299174070358276, "sampling/importance_sampling_ratio/mean": 1.0169990062713623, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7186378911137581, "clip_ratio/low_mean": 0.023024425841867924, "clip_ratio/low_min": 0.023024425841867924, "clip_ratio/high_mean": 0.06399360997602344, "clip_ratio/high_max": 0.06399360997602344, "clip_ratio/region_mean": 0.08701803581789136, "reward_total_mean": 0.45846500992774963, "reward_meter_mean": 0.8800153732299805, "reward_meter_std": 0.16372206807136536, "reward_count_adherence_mean": 0.5, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.788138747215271, "reward_repeat_soft_std": 0.09071363508701324, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.45846500992774963, "reward_total_composite_std": 0.04814787209033966} {"timestamp_utc": "2026-04-13T11:49:09Z", "mode": "train", "global_step": 1817, "epoch": 0.18252134605725767, "loss": -0.0054, "grad_norm": 6.091757297515869, "learning_rate": 4.496969696969697e-06, "num_tokens": 3255702.0, "completions/mean_length": 101.25, "completions/min_length": 79.0, "completions/max_length": 119.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 101.25, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 119.0, "rewards/meter/mean": 0.931243360042572, "rewards/meter/std": 0.1261349767446518, "rewards/count_adherence/mean": 0.5416666865348816, "rewards/count_adherence/std": 0.17251639068126678, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8289350271224976, "rewards/repeat_soft/std": 0.03682198002934456, "rewards/judge_quality/mean": 0.6487500071525574, "rewards/judge_quality/std": 0.2456151396036148, "rewards/total_composite/mean": 0.6251682043075562, "rewards/total_composite/std": 0.15108372271060944, "reward": 0.6251682043075562, "reward_std": 0.15108372271060944, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09348579496145248, "sampling/sampling_logp_difference/max": 1.7902698516845703, "sampling/importance_sampling_ratio/min": 0.16691511869430542, "sampling/importance_sampling_ratio/mean": 1.0175946950912476, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6182441972196102, "clip_ratio/low_mean": 0.0432027718052268, "clip_ratio/low_min": 0.0432027718052268, "clip_ratio/high_mean": 0.026850812137126923, "clip_ratio/high_max": 0.026850812137126923, "clip_ratio/region_mean": 0.07005358394235373, "reward_total_mean": 0.6251682043075562, "reward_meter_mean": 0.931243360042572, "reward_meter_std": 0.1261349767446518, "reward_count_adherence_mean": 0.5416666865348816, "reward_count_adherence_std": 0.17251639068126678, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8289350271224976, "reward_repeat_soft_std": 0.03682198002934456, "reward_judge_quality_mean": 0.6487500071525574, "reward_judge_quality_std": 0.2456151396036148, "reward_total_composite_mean": 0.6251682043075562, "reward_total_composite_std": 0.15108372271060944} {"timestamp_utc": "2026-04-13T11:49:15Z", "mode": "train", "global_step": 1818, "epoch": 0.18262179809141135, "loss": 0.1082, "grad_norm": 9.090434074401855, "learning_rate": 4.493939393939395e-06, "num_tokens": 3257463.0, "completions/mean_length": 49.125, "completions/min_length": 38.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.125, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.76106858253479, "rewards/meter/std": 0.3347349464893341, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.26726123690605164, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9026095867156982, "rewards/repeat_soft/std": 0.061921849846839905, "rewards/judge_quality/mean": 0.46875, "rewards/judge_quality/std": 0.10260014235973358, "rewards/total_composite/mean": 0.5050399303436279, "rewards/total_composite/std": 0.09929977357387543, "reward": 0.5050399303436279, "reward_std": 0.09929978847503662, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08518665283918381, "sampling/sampling_logp_difference/max": 1.5348838567733765, "sampling/importance_sampling_ratio/min": 0.2154807299375534, "sampling/importance_sampling_ratio/mean": 1.012081503868103, "sampling/importance_sampling_ratio/max": 1.8807759284973145, "entropy": 0.6251075603067875, "clip_ratio/low_mean": 0.043447833973914385, "clip_ratio/low_min": 0.043447833973914385, "clip_ratio/high_mean": 0.03728580195456743, "clip_ratio/high_max": 0.03728580195456743, "clip_ratio/region_mean": 0.08073363592848182, "reward_total_mean": 0.5050399303436279, "reward_meter_mean": 0.76106858253479, "reward_meter_std": 0.3347349464893341, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.26726123690605164, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9026095867156982, "reward_repeat_soft_std": 0.061921849846839905, "reward_judge_quality_mean": 0.46875, "reward_judge_quality_std": 0.10260014235973358, "reward_total_composite_mean": 0.5050399303436279, "reward_total_composite_std": 0.09929977357387543} {"timestamp_utc": "2026-04-13T11:49:21Z", "mode": "train", "global_step": 1819, "epoch": 0.18272225012556503, "loss": -0.0971, "grad_norm": 11.203838348388672, "learning_rate": 4.490909090909091e-06, "num_tokens": 3259080.0, "completions/mean_length": 41.125, "completions/min_length": 33.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.125, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.6667453646659851, "rewards/meter/std": 0.3940989077091217, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9571197628974915, "rewards/repeat_soft/std": 0.044309817254543304, "rewards/judge_quality/mean": 0.5062499642372131, "rewards/judge_quality/std": 0.14201989769935608, "rewards/total_composite/mean": 0.57743239402771, "rewards/total_composite/std": 0.16943275928497314, "reward": 0.57743239402771, "reward_std": 0.16943275928497314, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13686740398406982, "sampling/sampling_logp_difference/max": 1.297973394393921, "sampling/importance_sampling_ratio/min": 0.29697665572166443, "sampling/importance_sampling_ratio/mean": 1.0200828313827515, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9041113927960396, "clip_ratio/low_mean": 0.05710151046514511, "clip_ratio/low_min": 0.05710151046514511, "clip_ratio/high_mean": 0.09784480463713408, "clip_ratio/high_max": 0.09784480463713408, "clip_ratio/region_mean": 0.1549463151022792, "reward_total_mean": 0.57743239402771, "reward_meter_mean": 0.6667453646659851, "reward_meter_std": 0.3940989077091217, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9571197628974915, "reward_repeat_soft_std": 0.044309817254543304, "reward_judge_quality_mean": 0.5062499642372131, "reward_judge_quality_std": 0.14201989769935608, "reward_total_composite_mean": 0.57743239402771, "reward_total_composite_std": 0.16943275928497314} {"timestamp_utc": "2026-04-13T11:49:28Z", "mode": "train", "global_step": 1820, "epoch": 0.18282270215971874, "loss": 0.0264, "grad_norm": 12.091035842895508, "learning_rate": 4.487878787878788e-06, "num_tokens": 3260612.0, "completions/mean_length": 35.5, "completions/min_length": 29.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.5, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.5414127707481384, "rewards/meter/std": 0.3649313747882843, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9394425749778748, "rewards/repeat_soft/std": 0.05158919468522072, "rewards/judge_quality/mean": 0.6150000095367432, "rewards/judge_quality/std": 0.25286927819252014, "rewards/total_composite/mean": 0.5937775373458862, "rewards/total_composite/std": 0.23556756973266602, "reward": 0.5937775373458862, "reward_std": 0.23556756973266602, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10851365327835083, "sampling/sampling_logp_difference/max": 1.3843728303909302, "sampling/importance_sampling_ratio/min": 0.3599967360496521, "sampling/importance_sampling_ratio/mean": 1.012622594833374, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5805025473237038, "clip_ratio/low_mean": 0.05828921776264906, "clip_ratio/low_min": 0.05828921776264906, "clip_ratio/high_mean": 0.028174937702715397, "clip_ratio/high_max": 0.028174937702715397, "clip_ratio/region_mean": 0.08646415546536446, "reward_total_mean": 0.5937775373458862, "reward_meter_mean": 0.5414127707481384, "reward_meter_std": 0.3649313747882843, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9394425749778748, "reward_repeat_soft_std": 0.05158919468522072, "reward_judge_quality_mean": 0.6150000095367432, "reward_judge_quality_std": 0.25286927819252014, "reward_total_composite_mean": 0.5937775373458862, "reward_total_composite_std": 0.23556756973266602} {"timestamp_utc": "2026-04-13T11:49:34Z", "mode": "train", "global_step": 1821, "epoch": 0.18292315419387242, "loss": 0.0455, "grad_norm": 17.626693725585938, "learning_rate": 4.4848484848484855e-06, "num_tokens": 3261961.0, "completions/mean_length": 20.625, "completions/min_length": 18.0, "completions/max_length": 24.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 20.625, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 24.0, "rewards/meter/mean": 0.5819944739341736, "rewards/meter/std": 0.4677361249923706, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.5055780410766602, "rewards/total_composite/std": 0.12993228435516357, "reward": 0.5055780410766602, "reward_std": 0.12993228435516357, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1103300154209137, "sampling/sampling_logp_difference/max": 1.2852458953857422, "sampling/importance_sampling_ratio/min": 0.2765825688838959, "sampling/importance_sampling_ratio/mean": 1.0284539461135864, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0949403643608093, "clip_ratio/low_mean": 0.018483709543943405, "clip_ratio/low_min": 0.018483709543943405, "clip_ratio/high_mean": 0.09154396597296, "clip_ratio/high_max": 0.09154396597296, "clip_ratio/region_mean": 0.1100276755169034, "reward_total_mean": 0.5055780410766602, "reward_meter_mean": 0.5819944739341736, "reward_meter_std": 0.4677361249923706, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.5055780410766602, "reward_total_composite_std": 0.12993228435516357} {"timestamp_utc": "2026-04-13T11:49:41Z", "mode": "train", "global_step": 1822, "epoch": 0.18302360622802613, "loss": 0.035, "grad_norm": 5.529888153076172, "learning_rate": 4.481818181818182e-06, "num_tokens": 3264335.0, "completions/mean_length": 130.75, "completions/min_length": 109.0, "completions/max_length": 160.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 130.75, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 160.0, "rewards/meter/mean": 0.4340297281742096, "rewards/meter/std": 0.22348229587078094, "rewards/count_adherence/mean": 0.4375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8360930681228638, "rewards/repeat_soft/std": 0.02245735377073288, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.331297367811203, "rewards/total_composite/std": 0.037000950425863266, "reward": 0.331297367811203, "reward_std": 0.03700094297528267, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10386641323566437, "sampling/sampling_logp_difference/max": 1.4150023460388184, "sampling/importance_sampling_ratio/min": 0.24292504787445068, "sampling/importance_sampling_ratio/mean": 1.0231037139892578, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8596381545066833, "clip_ratio/low_mean": 0.05849697347730398, "clip_ratio/low_min": 0.05849697347730398, "clip_ratio/high_mean": 0.03997781686484814, "clip_ratio/high_max": 0.03997781686484814, "clip_ratio/region_mean": 0.09847479034215212, "reward_total_mean": 0.331297367811203, "reward_meter_mean": 0.4340297281742096, "reward_meter_std": 0.22348229587078094, "reward_count_adherence_mean": 0.4375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8360930681228638, "reward_repeat_soft_std": 0.02245735377073288, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.331297367811203, "reward_total_composite_std": 0.037000950425863266} {"timestamp_utc": "2026-04-13T11:49:53Z", "mode": "train", "global_step": 1823, "epoch": 0.1831240582621798, "loss": -0.0888, "grad_norm": 2.9576637744903564, "learning_rate": 4.478787878787879e-06, "num_tokens": 3265815.0, "completions/mean_length": 97.0, "completions/min_length": 33.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 37.71428680419922, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.6791916489601135, "rewards/meter/std": 0.44092419743537903, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.904679536819458, "rewards/repeat_soft/std": 0.05452427640557289, "rewards/judge_quality/mean": 0.5762500166893005, "rewards/judge_quality/std": 0.31513887643814087, "rewards/total_composite/mean": 0.6027641892433167, "rewards/total_composite/std": 0.3282713294029236, "reward": 0.6027641892433167, "reward_std": 0.3282712996006012, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10675515979528427, "sampling/sampling_logp_difference/max": 1.2832794189453125, "sampling/importance_sampling_ratio/min": 0.2771269977092743, "sampling/importance_sampling_ratio/mean": 1.016241431236267, "sampling/importance_sampling_ratio/max": 1.4655214548110962, "entropy": 0.7538627684116364, "clip_ratio/low_mean": 0.020707071293145418, "clip_ratio/low_min": 0.020707071293145418, "clip_ratio/high_mean": 0.058742116671055555, "clip_ratio/high_max": 0.058742116671055555, "clip_ratio/region_mean": 0.07944918796420097, "reward_total_mean": 0.6027641892433167, "reward_meter_mean": 0.6791916489601135, "reward_meter_std": 0.44092419743537903, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.904679536819458, "reward_repeat_soft_std": 0.05452427640557289, "reward_judge_quality_mean": 0.5762500166893005, "reward_judge_quality_std": 0.31513887643814087, "reward_total_composite_mean": 0.6027641892433167, "reward_total_composite_std": 0.3282713294029236} {"timestamp_utc": "2026-04-13T11:49:59Z", "mode": "train", "global_step": 1824, "epoch": 0.1832245102963335, "loss": 0.0017, "grad_norm": 18.037513732910156, "learning_rate": 4.4757575757575765e-06, "num_tokens": 3267181.0, "completions/mean_length": 20.75, "completions/min_length": 19.0, "completions/max_length": 24.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 20.75, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 24.0, "rewards/meter/mean": 0.8389955163002014, "rewards/meter/std": 0.2952535152435303, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9601136445999146, "rewards/repeat_soft/std": 0.006749649066478014, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.5754857063293457, "rewards/total_composite/std": 0.08214115351438522, "reward": 0.5754857063293457, "reward_std": 0.08214113116264343, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13900591433048248, "sampling/sampling_logp_difference/max": 1.2857437133789062, "sampling/importance_sampling_ratio/min": 0.2764449119567871, "sampling/importance_sampling_ratio/mean": 1.0151656866073608, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9926154911518097, "clip_ratio/low_mean": 0.03157894778996706, "clip_ratio/low_min": 0.03157894778996706, "clip_ratio/high_mean": 0.08142809476703405, "clip_ratio/high_max": 0.08142809476703405, "clip_ratio/region_mean": 0.11300704255700111, "reward_total_mean": 0.5754857063293457, "reward_meter_mean": 0.8389955163002014, "reward_meter_std": 0.2952535152435303, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9601136445999146, "reward_repeat_soft_std": 0.006749649066478014, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.5754857063293457, "reward_total_composite_std": 0.08214115351438522} {"timestamp_utc": "2026-04-13T11:50:05Z", "mode": "train", "global_step": 1825, "epoch": 0.1833249623304872, "loss": -0.0043, "grad_norm": 12.696378707885742, "learning_rate": 4.472727272727273e-06, "num_tokens": 3268846.0, "completions/mean_length": 43.125, "completions/min_length": 33.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.125, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.650660514831543, "rewards/meter/std": 0.35668233036994934, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9241976141929626, "rewards/repeat_soft/std": 0.05137051269412041, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.5163896083831787, "rewards/total_composite/std": 0.09794284403324127, "reward": 0.5163896083831787, "reward_std": 0.09794283658266068, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11815313994884491, "sampling/sampling_logp_difference/max": 1.5346651077270508, "sampling/importance_sampling_ratio/min": 0.21552786231040955, "sampling/importance_sampling_ratio/mean": 1.007631540298462, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8516337424516678, "clip_ratio/low_mean": 0.044034091755747795, "clip_ratio/low_min": 0.044034091755747795, "clip_ratio/high_mean": 0.051652208901941776, "clip_ratio/high_max": 0.051652208901941776, "clip_ratio/region_mean": 0.09568630065768957, "reward_total_mean": 0.5163896083831787, "reward_meter_mean": 0.650660514831543, "reward_meter_std": 0.35668233036994934, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9241976141929626, "reward_repeat_soft_std": 0.05137051269412041, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.5163896083831787, "reward_total_composite_std": 0.09794284403324127} {"timestamp_utc": "2026-04-13T11:50:13Z", "mode": "train", "global_step": 1826, "epoch": 0.18342541436464088, "loss": 0.0322, "grad_norm": 5.049038410186768, "learning_rate": 4.46969696969697e-06, "num_tokens": 3271207.0, "completions/mean_length": 140.125, "completions/min_length": 96.0, "completions/max_length": 199.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 140.125, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 199.0, "rewards/meter/mean": 0.8653736114501953, "rewards/meter/std": 0.15277306735515594, "rewards/count_adherence/mean": 0.40625, "rewards/count_adherence/std": 0.18600596487522125, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6539048552513123, "rewards/repeat_soft/std": 0.17207203805446625, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.45973342657089233, "rewards/total_composite/std": 0.1313636749982834, "reward": 0.45973342657089233, "reward_std": 0.1313636600971222, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08705134689807892, "sampling/sampling_logp_difference/max": 1.5911340713500977, "sampling/importance_sampling_ratio/min": 0.20369447767734528, "sampling/importance_sampling_ratio/mean": 1.0095009803771973, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6451926082372665, "clip_ratio/low_mean": 0.038877638056874275, "clip_ratio/low_min": 0.038877638056874275, "clip_ratio/high_mean": 0.04204982612282038, "clip_ratio/high_max": 0.04204982612282038, "clip_ratio/region_mean": 0.08092746417969465, "reward_total_mean": 0.45973342657089233, "reward_meter_mean": 0.8653736114501953, "reward_meter_std": 0.15277306735515594, "reward_count_adherence_mean": 0.40625, "reward_count_adherence_std": 0.18600596487522125, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6539048552513123, "reward_repeat_soft_std": 0.17207203805446625, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.45973342657089233, "reward_total_composite_std": 0.1313636749982834} {"timestamp_utc": "2026-04-13T11:50:26Z", "mode": "train", "global_step": 1827, "epoch": 0.18352586639879456, "loss": -0.0921, "grad_norm": 3.373699426651001, "learning_rate": 4.4666666666666665e-06, "num_tokens": 3272922.0, "completions/mean_length": 129.375, "completions/min_length": 61.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 74.71428680419922, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.6079522371292114, "rewards/meter/std": 0.27880096435546875, "rewards/count_adherence/mean": 0.4583333432674408, "rewards/count_adherence/std": 0.17251639068126678, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8452585935592651, "rewards/repeat_soft/std": 0.12623430788516998, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.23445606231689453, "rewards/total_composite/mean": 0.39055413007736206, "rewards/total_composite/std": 0.2326713651418686, "reward": 0.39055413007736206, "reward_std": 0.23267138004302979, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13240386545658112, "sampling/sampling_logp_difference/max": 1.4575583934783936, "sampling/importance_sampling_ratio/min": 0.23280400037765503, "sampling/importance_sampling_ratio/mean": 1.0285807847976685, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0534615963697433, "clip_ratio/low_mean": 0.04445145092904568, "clip_ratio/low_min": 0.04445145092904568, "clip_ratio/high_mean": 0.0673421062529087, "clip_ratio/high_max": 0.0673421062529087, "clip_ratio/region_mean": 0.11179355718195438, "reward_total_mean": 0.39055413007736206, "reward_meter_mean": 0.6079522371292114, "reward_meter_std": 0.27880096435546875, "reward_count_adherence_mean": 0.4583333432674408, "reward_count_adherence_std": 0.17251639068126678, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8452585935592651, "reward_repeat_soft_std": 0.12623430788516998, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.23445606231689453, "reward_total_composite_mean": 0.39055413007736206, "reward_total_composite_std": 0.2326713651418686} {"timestamp_utc": "2026-04-13T11:50:31Z", "mode": "train", "global_step": 1828, "epoch": 0.18362631843294827, "loss": -0.0703, "grad_norm": 16.006153106689453, "learning_rate": 4.463636363636364e-06, "num_tokens": 3274493.0, "completions/mean_length": 36.375, "completions/min_length": 28.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.375, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.7035261392593384, "rewards/meter/std": 0.3472166955471039, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9747592806816101, "rewards/repeat_soft/std": 0.03407462313771248, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.5427587032318115, "rewards/total_composite/std": 0.10208763182163239, "reward": 0.5427587032318115, "reward_std": 0.10208763182163239, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11395373940467834, "sampling/sampling_logp_difference/max": 1.2686762809753418, "sampling/importance_sampling_ratio/min": 0.28120359778404236, "sampling/importance_sampling_ratio/mean": 0.9948596358299255, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7612007334828377, "clip_ratio/low_mean": 0.02616995107382536, "clip_ratio/low_min": 0.02616995107382536, "clip_ratio/high_mean": 0.08919057063758373, "clip_ratio/high_max": 0.08919057063758373, "clip_ratio/region_mean": 0.11536052171140909, "reward_total_mean": 0.5427587032318115, "reward_meter_mean": 0.7035261392593384, "reward_meter_std": 0.3472166955471039, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9747592806816101, "reward_repeat_soft_std": 0.03407462313771248, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.5427587032318115, "reward_total_composite_std": 0.10208763182163239} {"timestamp_utc": "2026-04-13T11:50:39Z", "mode": "train", "global_step": 1829, "epoch": 0.18372677046710195, "loss": 0.1141, "grad_norm": 5.330620288848877, "learning_rate": 4.460606060606061e-06, "num_tokens": 3276974.0, "completions/mean_length": 147.125, "completions/min_length": 98.0, "completions/max_length": 193.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 147.125, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 193.0, "rewards/meter/mean": 0.8941610455513, "rewards/meter/std": 0.08215868473052979, "rewards/count_adherence/mean": 0.5625, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9671491384506226, "rewards/repeat_soft/std": 0.022209011018276215, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5016783475875854, "rewards/total_composite/std": 0.02898823283612728, "reward": 0.5016783475875854, "reward_std": 0.028988230973482132, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14371787011623383, "sampling/sampling_logp_difference/max": 1.7989435195922852, "sampling/importance_sampling_ratio/min": 0.16547362506389618, "sampling/importance_sampling_ratio/mean": 1.0169726610183716, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2326049730181694, "clip_ratio/low_mean": 0.08393213618546724, "clip_ratio/low_min": 0.08393213618546724, "clip_ratio/high_mean": 0.059747302904725075, "clip_ratio/high_max": 0.059747302904725075, "clip_ratio/region_mean": 0.14367943909019232, "reward_total_mean": 0.5016783475875854, "reward_meter_mean": 0.8941610455513, "reward_meter_std": 0.08215868473052979, "reward_count_adherence_mean": 0.5625, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9671491384506226, "reward_repeat_soft_std": 0.022209011018276215, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5016783475875854, "reward_total_composite_std": 0.02898823283612728} {"timestamp_utc": "2026-04-13T11:50:47Z", "mode": "train", "global_step": 1830, "epoch": 0.18382722250125566, "loss": 0.0691, "grad_norm": 5.620277404785156, "learning_rate": 4.4575757575757575e-06, "num_tokens": 3279432.0, "completions/mean_length": 136.25, "completions/min_length": 114.0, "completions/max_length": 162.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 136.25, "completions/min_terminated_length": 114.0, "completions/max_terminated_length": 162.0, "rewards/meter/mean": 0.7948927879333496, "rewards/meter/std": 0.18234854936599731, "rewards/count_adherence/mean": 0.5, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.835483968257904, "rewards/repeat_soft/std": 0.0281214602291584, "rewards/judge_quality/mean": 0.4950000047683716, "rewards/judge_quality/std": 0.13887304067611694, "rewards/total_composite/mean": 0.48247966170310974, "rewards/total_composite/std": 0.09574178606271744, "reward": 0.48247966170310974, "reward_std": 0.09574179351329803, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09292007237672806, "sampling/sampling_logp_difference/max": 1.4724364280700684, "sampling/importance_sampling_ratio/min": 0.2293659746646881, "sampling/importance_sampling_ratio/mean": 1.0040966272354126, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7820024564862251, "clip_ratio/low_mean": 0.041064705699682236, "clip_ratio/low_min": 0.041064705699682236, "clip_ratio/high_mean": 0.031241677701473236, "clip_ratio/high_max": 0.031241677701473236, "clip_ratio/region_mean": 0.07230638340115547, "reward_total_mean": 0.48247966170310974, "reward_meter_mean": 0.7948927879333496, "reward_meter_std": 0.18234854936599731, "reward_count_adherence_mean": 0.5, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.835483968257904, "reward_repeat_soft_std": 0.0281214602291584, "reward_judge_quality_mean": 0.4950000047683716, "reward_judge_quality_std": 0.13887304067611694, "reward_total_composite_mean": 0.48247966170310974, "reward_total_composite_std": 0.09574178606271744} {"timestamp_utc": "2026-04-13T11:50:54Z", "mode": "train", "global_step": 1831, "epoch": 0.18392767453540934, "loss": -0.0368, "grad_norm": 8.392786979675293, "learning_rate": 4.454545454545455e-06, "num_tokens": 3281491.0, "completions/mean_length": 86.375, "completions/min_length": 72.0, "completions/max_length": 110.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.375, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 110.0, "rewards/meter/mean": 0.8850648999214172, "rewards/meter/std": 0.1256731152534485, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8781824111938477, "rewards/repeat_soft/std": 0.02163693681359291, "rewards/judge_quality/mean": 0.5824999809265137, "rewards/judge_quality/std": 0.23260943591594696, "rewards/total_composite/mean": 0.603774905204773, "rewards/total_composite/std": 0.14991037547588348, "reward": 0.603774905204773, "reward_std": 0.14991037547588348, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1255534440279007, "sampling/sampling_logp_difference/max": 1.7929372787475586, "sampling/importance_sampling_ratio/min": 0.166470468044281, "sampling/importance_sampling_ratio/mean": 1.0084514617919922, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9133397415280342, "clip_ratio/low_mean": 0.07230066228657961, "clip_ratio/low_min": 0.07230066228657961, "clip_ratio/high_mean": 0.05087065976113081, "clip_ratio/high_max": 0.05087065976113081, "clip_ratio/region_mean": 0.12317132204771042, "reward_total_mean": 0.603774905204773, "reward_meter_mean": 0.8850648999214172, "reward_meter_std": 0.1256731152534485, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8781824111938477, "reward_repeat_soft_std": 0.02163693681359291, "reward_judge_quality_mean": 0.5824999809265137, "reward_judge_quality_std": 0.23260943591594696, "reward_total_composite_mean": 0.603774905204773, "reward_total_composite_std": 0.14991037547588348} {"timestamp_utc": "2026-04-13T11:51:06Z", "mode": "train", "global_step": 1832, "epoch": 0.18402812656956302, "loss": -0.2005, "grad_norm": 1.9224159717559814, "learning_rate": 4.451515151515152e-06, "num_tokens": 3284013.0, "completions/mean_length": 177.25, "completions/min_length": 119.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 129.42857360839844, "completions/min_terminated_length": 119.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.812292218208313, "rewards/meter/std": 0.18085992336273193, "rewards/count_adherence/mean": 0.5625, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8280702233314514, "rewards/repeat_soft/std": 0.0513884536921978, "rewards/judge_quality/mean": 0.6700000166893005, "rewards/judge_quality/std": 0.27255406975746155, "rewards/total_composite/mean": 0.5601329803466797, "rewards/total_composite/std": 0.2452675700187683, "reward": 0.5601329803466797, "reward_std": 0.24526755511760712, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0855712890625, "sampling/sampling_logp_difference/max": 1.4115657806396484, "sampling/importance_sampling_ratio/min": 0.2437613159418106, "sampling/importance_sampling_ratio/mean": 1.0131275653839111, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5874464139342308, "clip_ratio/low_mean": 0.010576923377811909, "clip_ratio/low_min": 0.010576923377811909, "clip_ratio/high_mean": 0.06200599577277899, "clip_ratio/high_max": 0.06200599577277899, "clip_ratio/region_mean": 0.0725829191505909, "reward_total_mean": 0.5601329803466797, "reward_meter_mean": 0.812292218208313, "reward_meter_std": 0.18085992336273193, "reward_count_adherence_mean": 0.5625, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8280702233314514, "reward_repeat_soft_std": 0.0513884536921978, "reward_judge_quality_mean": 0.6700000166893005, "reward_judge_quality_std": 0.27255406975746155, "reward_total_composite_mean": 0.5601329803466797, "reward_total_composite_std": 0.2452675700187683} {"timestamp_utc": "2026-04-13T11:51:14Z", "mode": "train", "global_step": 1833, "epoch": 0.18412857860371673, "loss": 0.0703, "grad_norm": 5.307923316955566, "learning_rate": 4.448484848484848e-06, "num_tokens": 3286692.0, "completions/mean_length": 160.875, "completions/min_length": 129.0, "completions/max_length": 179.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 160.875, "completions/min_terminated_length": 129.0, "completions/max_terminated_length": 179.0, "rewards/meter/mean": 0.8338100910186768, "rewards/meter/std": 0.20722128450870514, "rewards/count_adherence/mean": 0.5625, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7941572070121765, "rewards/repeat_soft/std": 0.034432124346494675, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.45925372838974, "rewards/total_composite/std": 0.06867709010839462, "reward": 0.45925372838974, "reward_std": 0.06867708265781403, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10376652330160141, "sampling/sampling_logp_difference/max": 1.7389531135559082, "sampling/importance_sampling_ratio/min": 0.1757042557001114, "sampling/importance_sampling_ratio/mean": 1.016282558441162, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9125862717628479, "clip_ratio/low_mean": 0.031769067980349064, "clip_ratio/low_min": 0.031769067980349064, "clip_ratio/high_mean": 0.04758817981928587, "clip_ratio/high_max": 0.04758817981928587, "clip_ratio/region_mean": 0.07935724779963493, "reward_total_mean": 0.45925372838974, "reward_meter_mean": 0.8338100910186768, "reward_meter_std": 0.20722128450870514, "reward_count_adherence_mean": 0.5625, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7941572070121765, "reward_repeat_soft_std": 0.034432124346494675, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.45925372838974, "reward_total_composite_std": 0.06867709010839462} {"timestamp_utc": "2026-04-13T11:51:20Z", "mode": "train", "global_step": 1834, "epoch": 0.1842290306378704, "loss": -0.0049, "grad_norm": 11.121776580810547, "learning_rate": 4.445454545454546e-06, "num_tokens": 3288381.0, "completions/mean_length": 48.125, "completions/min_length": 43.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.125, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.8402247428894043, "rewards/meter/std": 0.3212410509586334, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.99656742811203, "rewards/repeat_soft/std": 0.005215909332036972, "rewards/judge_quality/mean": 0.5024999976158142, "rewards/judge_quality/std": 0.1348809152841568, "rewards/total_composite/mean": 0.6299073696136475, "rewards/total_composite/std": 0.14270322024822235, "reward": 0.6299073696136475, "reward_std": 0.14270322024822235, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.143593430519104, "sampling/sampling_logp_difference/max": 1.4234967231750488, "sampling/importance_sampling_ratio/min": 0.24087028205394745, "sampling/importance_sampling_ratio/mean": 1.0094298124313354, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3203355967998505, "clip_ratio/low_mean": 0.07586692273616791, "clip_ratio/low_min": 0.07586692273616791, "clip_ratio/high_mean": 0.05965951643884182, "clip_ratio/high_max": 0.05965951643884182, "clip_ratio/region_mean": 0.13552643917500973, "reward_total_mean": 0.6299073696136475, "reward_meter_mean": 0.8402247428894043, "reward_meter_std": 0.3212410509586334, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.99656742811203, "reward_repeat_soft_std": 0.005215909332036972, "reward_judge_quality_mean": 0.5024999976158142, "reward_judge_quality_std": 0.1348809152841568, "reward_total_composite_mean": 0.6299073696136475, "reward_total_composite_std": 0.14270322024822235} {"timestamp_utc": "2026-04-13T11:51:32Z", "mode": "train", "global_step": 1835, "epoch": 0.18432948267202412, "loss": -0.1881, "grad_norm": 1.684330701828003, "learning_rate": 4.442424242424243e-06, "num_tokens": 3291010.0, "completions/mean_length": 278.625, "completions/min_length": 149.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 200.83334350585938, "completions/min_terminated_length": 149.0, "completions/max_terminated_length": 235.0, "rewards/meter/mean": 0.635834813117981, "rewards/meter/std": 0.24058298766613007, "rewards/count_adherence/mean": 0.5625, "rewards/count_adherence/std": 0.08625820279121399, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.7982053756713867, "rewards/repeat_soft/std": 0.06978677213191986, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.19427062571048737, "rewards/total_composite/mean": 0.34005433320999146, "rewards/total_composite/std": 0.16458089649677277, "reward": 0.34005433320999146, "reward_std": 0.16458088159561157, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09030982106924057, "sampling/sampling_logp_difference/max": 1.9321351051330566, "sampling/importance_sampling_ratio/min": 0.1961755007505417, "sampling/importance_sampling_ratio/mean": 1.0259186029434204, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5805070772767067, "clip_ratio/low_mean": 0.007575757801532745, "clip_ratio/low_min": 0.007575757801532745, "clip_ratio/high_mean": 0.053594366647303104, "clip_ratio/high_max": 0.053594366647303104, "clip_ratio/region_mean": 0.06117012444883585, "reward_total_mean": 0.34005433320999146, "reward_meter_mean": 0.635834813117981, "reward_meter_std": 0.24058298766613007, "reward_count_adherence_mean": 0.5625, "reward_count_adherence_std": 0.08625820279121399, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.7982053756713867, "reward_repeat_soft_std": 0.06978677213191986, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.19427062571048737, "reward_total_composite_mean": 0.34005433320999146, "reward_total_composite_std": 0.16458089649677277} {"timestamp_utc": "2026-04-13T11:51:38Z", "mode": "train", "global_step": 1836, "epoch": 0.1844299347061778, "loss": 0.0789, "grad_norm": 7.766142845153809, "learning_rate": 4.43939393939394e-06, "num_tokens": 3292684.0, "completions/mean_length": 47.25, "completions/min_length": 40.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.25, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.8402402400970459, "rewards/meter/std": 0.21419556438922882, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9650239944458008, "rewards/repeat_soft/std": 0.03594968095421791, "rewards/judge_quality/mean": 0.5562499761581421, "rewards/judge_quality/std": 0.2249404937028885, "rewards/total_composite/mean": 0.6383761167526245, "rewards/total_composite/std": 0.12639735639095306, "reward": 0.6383761167526245, "reward_std": 0.12639737129211426, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1247914731502533, "sampling/sampling_logp_difference/max": 1.0808696746826172, "sampling/importance_sampling_ratio/min": 0.3393003046512604, "sampling/importance_sampling_ratio/mean": 1.0129711627960205, "sampling/importance_sampling_ratio/max": 1.6747112274169922, "entropy": 0.9699152186512947, "clip_ratio/low_mean": 0.07616180088371038, "clip_ratio/low_min": 0.07616180088371038, "clip_ratio/high_mean": 0.009146341122686863, "clip_ratio/high_max": 0.009146341122686863, "clip_ratio/region_mean": 0.08530814200639725, "reward_total_mean": 0.6383761167526245, "reward_meter_mean": 0.8402402400970459, "reward_meter_std": 0.21419556438922882, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9650239944458008, "reward_repeat_soft_std": 0.03594968095421791, "reward_judge_quality_mean": 0.5562499761581421, "reward_judge_quality_std": 0.2249404937028885, "reward_total_composite_mean": 0.6383761167526245, "reward_total_composite_std": 0.12639735639095306} {"timestamp_utc": "2026-04-13T11:51:46Z", "mode": "train", "global_step": 1837, "epoch": 0.18453038674033148, "loss": 0.1382, "grad_norm": 6.6513566970825195, "learning_rate": 4.436363636363637e-06, "num_tokens": 3295231.0, "completions/mean_length": 126.375, "completions/min_length": 86.0, "completions/max_length": 161.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 126.375, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 161.0, "rewards/meter/mean": 0.801716685295105, "rewards/meter/std": 0.24808786809444427, "rewards/count_adherence/mean": 0.6875, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8516167402267456, "rewards/repeat_soft/std": 0.04234924167394638, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.4964723587036133, "rewards/total_composite/std": 0.07371412217617035, "reward": 0.4964723587036133, "reward_std": 0.07371412217617035, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09954273700714111, "sampling/sampling_logp_difference/max": 2.240025758743286, "sampling/importance_sampling_ratio/min": 0.10645575821399689, "sampling/importance_sampling_ratio/mean": 1.008533239364624, "sampling/importance_sampling_ratio/max": 1.911268711090088, "entropy": 0.7372321635484695, "clip_ratio/low_mean": 0.02968578739091754, "clip_ratio/low_min": 0.02968578739091754, "clip_ratio/high_mean": 0.07113707065582275, "clip_ratio/high_max": 0.07113707065582275, "clip_ratio/region_mean": 0.1008228580467403, "reward_total_mean": 0.4964723587036133, "reward_meter_mean": 0.801716685295105, "reward_meter_std": 0.24808786809444427, "reward_count_adherence_mean": 0.6875, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8516167402267456, "reward_repeat_soft_std": 0.04234924167394638, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.4964723587036133, "reward_total_composite_std": 0.07371412217617035} {"timestamp_utc": "2026-04-13T11:51:52Z", "mode": "train", "global_step": 1838, "epoch": 0.1846308387744852, "loss": -0.1061, "grad_norm": 12.099515914916992, "learning_rate": 4.433333333333334e-06, "num_tokens": 3296819.0, "completions/mean_length": 35.5, "completions/min_length": 31.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.5, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.6132403612136841, "rewards/meter/std": 0.41704481840133667, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9716784954071045, "rewards/repeat_soft/std": 0.03539866581559181, "rewards/judge_quality/mean": 0.5862500071525574, "rewards/judge_quality/std": 0.2822834253311157, "rewards/total_composite/mean": 0.5992453098297119, "rewards/total_composite/std": 0.20921917259693146, "reward": 0.5992453098297119, "reward_std": 0.20921917259693146, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11490785330533981, "sampling/sampling_logp_difference/max": 1.231729507446289, "sampling/importance_sampling_ratio/min": 0.2917875051498413, "sampling/importance_sampling_ratio/mean": 0.9885565638542175, "sampling/importance_sampling_ratio/max": 1.5996848344802856, "entropy": 0.64373379945755, "clip_ratio/low_mean": 0.046913184225559235, "clip_ratio/low_min": 0.046913184225559235, "clip_ratio/high_mean": 0.06361937709152699, "clip_ratio/high_max": 0.06361937709152699, "clip_ratio/region_mean": 0.11053256131708622, "reward_total_mean": 0.5992453098297119, "reward_meter_mean": 0.6132403612136841, "reward_meter_std": 0.41704481840133667, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9716784954071045, "reward_repeat_soft_std": 0.03539866581559181, "reward_judge_quality_mean": 0.5862500071525574, "reward_judge_quality_std": 0.2822834253311157, "reward_total_composite_mean": 0.5992453098297119, "reward_total_composite_std": 0.20921917259693146} {"timestamp_utc": "2026-04-13T11:51:58Z", "mode": "train", "global_step": 1839, "epoch": 0.18473129080863887, "loss": 0.0012, "grad_norm": 13.442041397094727, "learning_rate": 4.430303030303031e-06, "num_tokens": 3298290.0, "completions/mean_length": 39.875, "completions/min_length": 34.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.875, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9158987402915955, "rewards/meter/std": 0.17611320316791534, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8697051405906677, "rewards/repeat_soft/std": 0.03761998936533928, "rewards/judge_quality/mean": 0.5824999809265137, "rewards/judge_quality/std": 0.23260943591594696, "rewards/total_composite/mean": 0.6640946865081787, "rewards/total_composite/std": 0.1221679225564003, "reward": 0.6640946865081787, "reward_std": 0.1221679151058197, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11774171888828278, "sampling/sampling_logp_difference/max": 1.8575263023376465, "sampling/importance_sampling_ratio/min": 0.1560581922531128, "sampling/importance_sampling_ratio/mean": 1.017078161239624, "sampling/importance_sampling_ratio/max": 1.6088244915008545, "entropy": 0.8993172124028206, "clip_ratio/low_mean": 0.04582105437293649, "clip_ratio/low_min": 0.04582105437293649, "clip_ratio/high_mean": 0.019720497075468302, "clip_ratio/high_max": 0.019720497075468302, "clip_ratio/region_mean": 0.06554155144840479, "reward_total_mean": 0.6640946865081787, "reward_meter_mean": 0.9158987402915955, "reward_meter_std": 0.17611320316791534, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8697051405906677, "reward_repeat_soft_std": 0.03761998936533928, "reward_judge_quality_mean": 0.5824999809265137, "reward_judge_quality_std": 0.23260943591594696, "reward_total_composite_mean": 0.6640946865081787, "reward_total_composite_std": 0.1221679225564003} {"timestamp_utc": "2026-04-13T11:52:10Z", "mode": "train", "global_step": 1840, "epoch": 0.18483174284279258, "loss": -0.229, "grad_norm": 1.546162486076355, "learning_rate": 4.4272727272727275e-06, "num_tokens": 3301079.0, "completions/mean_length": 204.625, "completions/min_length": 134.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 160.71429443359375, "completions/min_terminated_length": 134.0, "completions/max_terminated_length": 207.0, "rewards/meter/mean": 0.9185236692428589, "rewards/meter/std": 0.1468735933303833, "rewards/count_adherence/mean": 0.7749999761581421, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.7914280891418457, "rewards/repeat_soft/std": 0.046963345259428024, "rewards/judge_quality/mean": 0.3824999928474426, "rewards/judge_quality/std": 0.1060660108923912, "rewards/total_composite/mean": 0.45604652166366577, "rewards/total_composite/std": 0.18849195539951324, "reward": 0.45604652166366577, "reward_std": 0.18849195539951324, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08804997056722641, "sampling/sampling_logp_difference/max": 2.1487064361572266, "sampling/importance_sampling_ratio/min": 0.11663493514060974, "sampling/importance_sampling_ratio/mean": 1.0134732723236084, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5453835353255272, "clip_ratio/low_mean": 0.006410256493836641, "clip_ratio/low_min": 0.006410256493836641, "clip_ratio/high_mean": 0.06665645446628332, "clip_ratio/high_max": 0.06665645446628332, "clip_ratio/region_mean": 0.07306671096011996, "reward_total_mean": 0.45604652166366577, "reward_meter_mean": 0.9185236692428589, "reward_meter_std": 0.1468735933303833, "reward_count_adherence_mean": 0.7749999761581421, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.7914280891418457, "reward_repeat_soft_std": 0.046963345259428024, "reward_judge_quality_mean": 0.3824999928474426, "reward_judge_quality_std": 0.1060660108923912, "reward_total_composite_mean": 0.45604652166366577, "reward_total_composite_std": 0.18849195539951324} {"timestamp_utc": "2026-04-13T11:52:16Z", "mode": "train", "global_step": 1841, "epoch": 0.18493219487694626, "loss": -0.0022, "grad_norm": 11.853466987609863, "learning_rate": 4.424242424242425e-06, "num_tokens": 3302539.0, "completions/mean_length": 33.5, "completions/min_length": 28.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.5, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.9478296041488647, "rewards/meter/std": 0.06476016342639923, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9152984023094177, "rewards/repeat_soft/std": 0.05110251531004906, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.640056848526001, "rewards/total_composite/std": 0.11371893435716629, "reward": 0.640056848526001, "reward_std": 0.1137189269065857, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1272178441286087, "sampling/sampling_logp_difference/max": 1.722355842590332, "sampling/importance_sampling_ratio/min": 0.1786447912454605, "sampling/importance_sampling_ratio/mean": 1.0080894231796265, "sampling/importance_sampling_ratio/max": 1.6141676902770996, "entropy": 0.9794385582208633, "clip_ratio/low_mean": 0.08459542435593903, "clip_ratio/low_min": 0.08459542435593903, "clip_ratio/high_mean": 0.007575757801532745, "clip_ratio/high_max": 0.007575757801532745, "clip_ratio/region_mean": 0.09217118215747178, "reward_total_mean": 0.640056848526001, "reward_meter_mean": 0.9478296041488647, "reward_meter_std": 0.06476016342639923, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9152984023094177, "reward_repeat_soft_std": 0.05110251531004906, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.640056848526001, "reward_total_composite_std": 0.11371893435716629} {"timestamp_utc": "2026-04-13T11:52:28Z", "mode": "train", "global_step": 1842, "epoch": 0.18503264691109994, "loss": -0.2157, "grad_norm": 1.6094402074813843, "learning_rate": 4.421212121212122e-06, "num_tokens": 3305073.0, "completions/mean_length": 206.75, "completions/min_length": 131.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 163.1428680419922, "completions/min_terminated_length": 131.0, "completions/max_terminated_length": 182.0, "rewards/meter/mean": 0.837173581123352, "rewards/meter/std": 0.2746213674545288, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.853219747543335, "rewards/repeat_soft/std": 0.07812196016311646, "rewards/judge_quality/mean": 0.4737499952316284, "rewards/judge_quality/std": 0.254611074924469, "rewards/total_composite/mean": 0.521868109703064, "rewards/total_composite/std": 0.2212262898683548, "reward": 0.521868109703064, "reward_std": 0.2212262898683548, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11141830682754517, "sampling/sampling_logp_difference/max": 1.8805809020996094, "sampling/importance_sampling_ratio/min": 0.15250149369239807, "sampling/importance_sampling_ratio/mean": 1.0141204595565796, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7322415187954903, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0984646612778306, "clip_ratio/high_max": 0.0984646612778306, "clip_ratio/region_mean": 0.0984646612778306, "reward_total_mean": 0.521868109703064, "reward_meter_mean": 0.837173581123352, "reward_meter_std": 0.2746213674545288, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.853219747543335, "reward_repeat_soft_std": 0.07812196016311646, "reward_judge_quality_mean": 0.4737499952316284, "reward_judge_quality_std": 0.254611074924469, "reward_total_composite_mean": 0.521868109703064, "reward_total_composite_std": 0.2212262898683548} {"timestamp_utc": "2026-04-13T11:52:35Z", "mode": "train", "global_step": 1843, "epoch": 0.18513309894525365, "loss": 0.0401, "grad_norm": 9.5546236038208, "learning_rate": 4.418181818181818e-06, "num_tokens": 3306753.0, "completions/mean_length": 34.0, "completions/min_length": 32.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.937764048576355, "rewards/meter/std": 0.04446383938193321, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8251459002494812, "rewards/repeat_soft/std": 0.07597748935222626, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.5821202993392944, "rewards/total_composite/std": 0.01404713373631239, "reward": 0.5821202993392944, "reward_std": 0.014047140255570412, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05521784722805023, "sampling/sampling_logp_difference/max": 1.006291151046753, "sampling/importance_sampling_ratio/min": 0.42414647340774536, "sampling/importance_sampling_ratio/mean": 1.0107085704803467, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.34332825243473053, "clip_ratio/low_mean": 0.024603174766525626, "clip_ratio/low_min": 0.024603174766525626, "clip_ratio/high_mean": 0.04203570680692792, "clip_ratio/high_max": 0.04203570680692792, "clip_ratio/region_mean": 0.06663888157345355, "reward_total_mean": 0.5821202993392944, "reward_meter_mean": 0.937764048576355, "reward_meter_std": 0.04446383938193321, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8251459002494812, "reward_repeat_soft_std": 0.07597748935222626, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.5821202993392944, "reward_total_composite_std": 0.01404713373631239} {"timestamp_utc": "2026-04-13T11:52:42Z", "mode": "train", "global_step": 1844, "epoch": 0.18523355097940733, "loss": 0.0632, "grad_norm": 6.3035569190979, "learning_rate": 4.415151515151516e-06, "num_tokens": 3309227.0, "completions/mean_length": 117.25, "completions/min_length": 89.0, "completions/max_length": 142.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 117.25, "completions/min_terminated_length": 89.0, "completions/max_terminated_length": 142.0, "rewards/meter/mean": 0.9056618809700012, "rewards/meter/std": 0.11452306807041168, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7821943759918213, "rewards/repeat_soft/std": 0.04733280465006828, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.5445857048034668, "rewards/total_composite/std": 0.07488405704498291, "reward": 0.5445857048034668, "reward_std": 0.07488404214382172, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08216746896505356, "sampling/sampling_logp_difference/max": 1.2165040969848633, "sampling/importance_sampling_ratio/min": 0.29626408219337463, "sampling/importance_sampling_ratio/mean": 1.0209424495697021, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6338837668299675, "clip_ratio/low_mean": 0.06343623623251915, "clip_ratio/low_min": 0.06343623623251915, "clip_ratio/high_mean": 0.01123595517128706, "clip_ratio/high_max": 0.01123595517128706, "clip_ratio/region_mean": 0.07467219140380621, "reward_total_mean": 0.5445857048034668, "reward_meter_mean": 0.9056618809700012, "reward_meter_std": 0.11452306807041168, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7821943759918213, "reward_repeat_soft_std": 0.04733280465006828, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.5445857048034668, "reward_total_composite_std": 0.07488405704498291} {"timestamp_utc": "2026-04-13T11:52:54Z", "mode": "train", "global_step": 1845, "epoch": 0.18533400301356104, "loss": -0.0873, "grad_norm": 3.8071818351745605, "learning_rate": 4.412121212121213e-06, "num_tokens": 3310756.0, "completions/mean_length": 97.125, "completions/min_length": 30.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 37.85714340209961, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.7326558232307434, "rewards/meter/std": 0.3586273789405823, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9280301928520203, "rewards/repeat_soft/std": 0.04183439910411835, "rewards/judge_quality/mean": 0.7487499713897705, "rewards/judge_quality/std": 0.332154780626297, "rewards/total_composite/mean": 0.7463199496269226, "rewards/total_composite/std": 0.22930696606636047, "reward": 0.7463199496269226, "reward_std": 0.22930695116519928, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12542758882045746, "sampling/sampling_logp_difference/max": 1.4823064804077148, "sampling/importance_sampling_ratio/min": 0.2271132618188858, "sampling/importance_sampling_ratio/mean": 1.0407088994979858, "sampling/importance_sampling_ratio/max": 1.644066572189331, "entropy": 0.9438460394740105, "clip_ratio/low_mean": 0.013095238711684942, "clip_ratio/low_min": 0.013095238711684942, "clip_ratio/high_mean": 0.03941104235127568, "clip_ratio/high_max": 0.03941104235127568, "clip_ratio/region_mean": 0.052506281062960625, "reward_total_mean": 0.7463199496269226, "reward_meter_mean": 0.7326558232307434, "reward_meter_std": 0.3586273789405823, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9280301928520203, "reward_repeat_soft_std": 0.04183439910411835, "reward_judge_quality_mean": 0.7487499713897705, "reward_judge_quality_std": 0.332154780626297, "reward_total_composite_mean": 0.7463199496269226, "reward_total_composite_std": 0.22930696606636047} {"timestamp_utc": "2026-04-13T11:53:06Z", "mode": "train", "global_step": 1846, "epoch": 0.18543445504771472, "loss": -0.2039, "grad_norm": 2.956387996673584, "learning_rate": 4.409090909090909e-06, "num_tokens": 3313146.0, "completions/mean_length": 162.75, "completions/min_length": 82.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 112.85714721679688, "completions/min_terminated_length": 82.0, "completions/max_terminated_length": 132.0, "rewards/meter/mean": 0.6697490215301514, "rewards/meter/std": 0.22099152207374573, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.7872285842895508, "rewards/repeat_soft/std": 0.07616162300109863, "rewards/judge_quality/mean": 0.6237499713897705, "rewards/judge_quality/std": 0.33907175064086914, "rewards/total_composite/mean": 0.5034215450286865, "rewards/total_composite/std": 0.2599618434906006, "reward": 0.5034215450286865, "reward_std": 0.2599618434906006, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08168012648820877, "sampling/sampling_logp_difference/max": 2.0867412090301514, "sampling/importance_sampling_ratio/min": 0.12409086525440216, "sampling/importance_sampling_ratio/mean": 1.0142415761947632, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.49081745743751526, "clip_ratio/low_mean": 0.01892037340439856, "clip_ratio/low_min": 0.01892037340439856, "clip_ratio/high_mean": 0.04009432205930352, "clip_ratio/high_max": 0.04009432205930352, "clip_ratio/region_mean": 0.05901469546370208, "reward_total_mean": 0.5034215450286865, "reward_meter_mean": 0.6697490215301514, "reward_meter_std": 0.22099152207374573, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.7872285842895508, "reward_repeat_soft_std": 0.07616162300109863, "reward_judge_quality_mean": 0.6237499713897705, "reward_judge_quality_std": 0.33907175064086914, "reward_total_composite_mean": 0.5034215450286865, "reward_total_composite_std": 0.2599618434906006} {"timestamp_utc": "2026-04-13T11:53:12Z", "mode": "train", "global_step": 1847, "epoch": 0.1855349070818684, "loss": 0.0214, "grad_norm": 5.1473188400268555, "learning_rate": 4.4060606060606066e-06, "num_tokens": 3315028.0, "completions/mean_length": 74.25, "completions/min_length": 60.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.25, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.8777709007263184, "rewards/meter/std": 0.18060046434402466, "rewards/count_adherence/mean": 0.7083333730697632, "rewards/count_adherence/std": 0.11785111576318741, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8241331577301025, "rewards/repeat_soft/std": 0.03560381755232811, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.5443388223648071, "rewards/total_composite/std": 0.12876400351524353, "reward": 0.5443388223648071, "reward_std": 0.12876400351524353, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05746019259095192, "sampling/sampling_logp_difference/max": 1.274916410446167, "sampling/importance_sampling_ratio/min": 0.2794543206691742, "sampling/importance_sampling_ratio/mean": 1.001806378364563, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.26211889274418354, "clip_ratio/low_mean": 0.023142279242165387, "clip_ratio/low_min": 0.023142279242165387, "clip_ratio/high_mean": 0.012660256586968899, "clip_ratio/high_max": 0.012660256586968899, "clip_ratio/region_mean": 0.035802535829134285, "reward_total_mean": 0.5443388223648071, "reward_meter_mean": 0.8777709007263184, "reward_meter_std": 0.18060046434402466, "reward_count_adherence_mean": 0.7083333730697632, "reward_count_adherence_std": 0.11785111576318741, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8241331577301025, "reward_repeat_soft_std": 0.03560381755232811, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.5443388223648071, "reward_total_composite_std": 0.12876400351524353} {"timestamp_utc": "2026-04-13T11:53:18Z", "mode": "train", "global_step": 1848, "epoch": 0.1856353591160221, "loss": -0.0234, "grad_norm": 13.755486488342285, "learning_rate": 4.403030303030304e-06, "num_tokens": 3316386.0, "completions/mean_length": 16.75, "completions/min_length": 15.0, "completions/max_length": 24.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 16.75, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 24.0, "rewards/meter/mean": 0.4936433434486389, "rewards/meter/std": 0.49729251861572266, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9505281448364258, "rewards/repeat_soft/std": 0.017931431531906128, "rewards/judge_quality/mean": 0.48624998331069946, "rewards/judge_quality/std": 0.1755755990743637, "rewards/total_composite/mean": 0.5175467133522034, "rewards/total_composite/std": 0.2085137963294983, "reward": 0.5175467133522034, "reward_std": 0.2085137963294983, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07921064645051956, "sampling/sampling_logp_difference/max": 0.8258018493652344, "sampling/importance_sampling_ratio/min": 0.43788373470306396, "sampling/importance_sampling_ratio/mean": 1.0038883686065674, "sampling/importance_sampling_ratio/max": 1.4596023559570312, "entropy": 0.6067915931344032, "clip_ratio/low_mean": 0.047222224064171314, "clip_ratio/low_min": 0.047222224064171314, "clip_ratio/high_mean": 0.045833335258066654, "clip_ratio/high_max": 0.045833335258066654, "clip_ratio/region_mean": 0.09305555932223797, "reward_total_mean": 0.5175467133522034, "reward_meter_mean": 0.4936433434486389, "reward_meter_std": 0.49729251861572266, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9505281448364258, "reward_repeat_soft_std": 0.017931431531906128, "reward_judge_quality_mean": 0.48624998331069946, "reward_judge_quality_std": 0.1755755990743637, "reward_total_composite_mean": 0.5175467133522034, "reward_total_composite_std": 0.2085137963294983} {"timestamp_utc": "2026-04-13T11:53:24Z", "mode": "train", "global_step": 1849, "epoch": 0.1857358111501758, "loss": 0.0543, "grad_norm": 11.871057510375977, "learning_rate": 4.4e-06, "num_tokens": 3318082.0, "completions/mean_length": 50.0, "completions/min_length": 37.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.0, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9819462299346924, "rewards/meter/std": 0.009022936224937439, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9770676493644714, "rewards/repeat_soft/std": 0.019794698804616928, "rewards/judge_quality/mean": 0.5862500071525574, "rewards/judge_quality/std": 0.22984081506729126, "rewards/total_composite/mean": 0.6959177255630493, "rewards/total_composite/std": 0.15842773020267487, "reward": 0.6959177255630493, "reward_std": 0.15842771530151367, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14248159527778625, "sampling/sampling_logp_difference/max": 1.2809314727783203, "sampling/importance_sampling_ratio/min": 0.2777784466743469, "sampling/importance_sampling_ratio/mean": 1.0329411029815674, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1133587881922722, "clip_ratio/low_mean": 0.0994670232757926, "clip_ratio/low_min": 0.0994670232757926, "clip_ratio/high_mean": 0.05312500149011612, "clip_ratio/high_max": 0.05312500149011612, "clip_ratio/region_mean": 0.15259202476590872, "reward_total_mean": 0.6959177255630493, "reward_meter_mean": 0.9819462299346924, "reward_meter_std": 0.009022936224937439, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9770676493644714, "reward_repeat_soft_std": 0.019794698804616928, "reward_judge_quality_mean": 0.5862500071525574, "reward_judge_quality_std": 0.22984081506729126, "reward_total_composite_mean": 0.6959177255630493, "reward_total_composite_std": 0.15842773020267487} {"timestamp_utc": "2026-04-13T11:53:36Z", "mode": "train", "global_step": 1850, "epoch": 0.18583626318432947, "loss": -0.2215, "grad_norm": 1.7200616598129272, "learning_rate": 4.3969696969696975e-06, "num_tokens": 3320144.0, "completions/mean_length": 225.75, "completions/min_length": 83.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 130.33334350585938, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 184.0, "rewards/meter/mean": 0.9502502679824829, "rewards/meter/std": 0.07333400100469589, "rewards/count_adherence/mean": 0.59375, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.8547098636627197, "rewards/repeat_soft/std": 0.04857774078845978, "rewards/judge_quality/mean": 0.3399999737739563, "rewards/judge_quality/std": 0.15052290260791779, "rewards/total_composite/mean": 0.3778553009033203, "rewards/total_composite/std": 0.2338632047176361, "reward": 0.3778553009033203, "reward_std": 0.2338632047176361, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11060090363025665, "sampling/sampling_logp_difference/max": 1.5649824142456055, "sampling/importance_sampling_ratio/min": 0.2090916782617569, "sampling/importance_sampling_ratio/mean": 1.007638692855835, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.569040559232235, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.07817493844777346, "clip_ratio/high_max": 0.07817493844777346, "clip_ratio/region_mean": 0.07817493844777346, "reward_total_mean": 0.3778553009033203, "reward_meter_mean": 0.9502502679824829, "reward_meter_std": 0.07333400100469589, "reward_count_adherence_mean": 0.59375, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.8547098636627197, "reward_repeat_soft_std": 0.04857774078845978, "reward_judge_quality_mean": 0.3399999737739563, "reward_judge_quality_std": 0.15052290260791779, "reward_total_composite_mean": 0.3778553009033203, "reward_total_composite_std": 0.2338632047176361} {"timestamp_utc": "2026-04-13T11:54:32Z", "mode": "eval", "global_step": 1850, "epoch": 0.18583626318432947, "eval_loss": NaN, "eval_runtime": 56.3018, "eval_samples_per_second": 1.421, "eval_steps_per_second": 0.178, "eval_num_tokens": 3320144.0, "eval_completions/mean_length": 102.0125, "eval_completions/min_length": 29.9, "eval_completions/max_length": 240.2, "eval_completions/clipped_ratio": 0.025, "eval_completions/mean_terminated_length": 91.89464416503907, "eval_completions/min_terminated_length": 29.9, "eval_completions/max_terminated_length": 170.3, "eval_rewards/meter/mean": 0.8248995125293732, "eval_rewards/meter/std": 0.23458105847239494, "eval_rewards/count_adherence/mean": 0.8335416734218597, "eval_rewards/count_adherence/std": 0.13577206507325174, "eval_rewards/hard_gate/mean": 0.975, "eval_rewards/hard_gate/std": 0.07071067690849304, "eval_rewards/repeat_soft/mean": 0.8762265861034393, "eval_rewards/repeat_soft/std": 0.09761511832475663, "eval_rewards/judge_quality/mean": 0.4736250102519989, "eval_rewards/judge_quality/std": 0.20385925192385912, "eval_rewards/total_composite/mean": 0.5502476394176483, "eval_rewards/total_composite/std": 0.1676765900105238, "eval_reward": 0.5502476394176483, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.06305473819375038, "eval_sampling/sampling_logp_difference/max": 1.197375774383545, "eval_sampling/importance_sampling_ratio/min": 0.31693738251924514, "eval_sampling/importance_sampling_ratio/mean": 1.0191887259483337, "eval_sampling/importance_sampling_ratio/max": 1.457780361175537, "eval_entropy": 0.6957984805107117, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5502476394176483, "eval_reward_meter_mean": 0.8248995125293732, "eval_reward_meter_std": 0.23458105847239494, "eval_reward_count_adherence_mean": 0.8335416734218597, "eval_reward_count_adherence_std": 0.13577206507325174, "eval_reward_hard_gate_mean": 0.975, "eval_reward_hard_gate_std": 0.07071067690849304, "eval_reward_repeat_soft_mean": 0.8762265861034393, "eval_reward_repeat_soft_std": 0.09761511832475663, "eval_reward_judge_quality_mean": 0.4736250102519989, "eval_reward_judge_quality_std": 0.20385925192385912, "eval_reward_total_composite_mean": 0.5502476394176483, "eval_reward_total_composite_std": 0.1676765900105238} {"timestamp_utc": "2026-04-13T11:54:42Z", "mode": "train", "global_step": 1851, "epoch": 0.18593671521848318, "loss": 0.0096, "grad_norm": 5.400481224060059, "learning_rate": 4.393939393939394e-06, "num_tokens": 3322238.0, "completions/mean_length": 95.75, "completions/min_length": 94.0, "completions/max_length": 101.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.75, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 101.0, "rewards/meter/mean": 0.9656529426574707, "rewards/meter/std": 0.008796771056950092, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7203428745269775, "rewards/repeat_soft/std": 0.05236854776740074, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.5083029270172119, "rewards/total_composite/std": 0.03721180930733681, "reward": 0.5083029270172119, "reward_std": 0.037211816757917404, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.057944923639297485, "sampling/sampling_logp_difference/max": 1.7273783683776855, "sampling/importance_sampling_ratio/min": 0.17774979770183563, "sampling/importance_sampling_ratio/mean": 1.0042781829833984, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.29475096240639687, "clip_ratio/low_mean": 0.007894736714661121, "clip_ratio/low_min": 0.007894736714661121, "clip_ratio/high_mean": 0.0548994611017406, "clip_ratio/high_max": 0.0548994611017406, "clip_ratio/region_mean": 0.06279419781640172, "reward_total_mean": 0.5083029270172119, "reward_meter_mean": 0.9656529426574707, "reward_meter_std": 0.008796771056950092, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7203428745269775, "reward_repeat_soft_std": 0.05236854776740074, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.5083029270172119, "reward_total_composite_std": 0.03721180930733681} {"timestamp_utc": "2026-04-13T11:54:48Z", "mode": "train", "global_step": 1852, "epoch": 0.18603716725263686, "loss": -0.0237, "grad_norm": 10.641679763793945, "learning_rate": 4.390909090909091e-06, "num_tokens": 3323871.0, "completions/mean_length": 41.125, "completions/min_length": 36.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.125, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.9228650331497192, "rewards/meter/std": 0.10433200746774673, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.895309329032898, "rewards/repeat_soft/std": 0.07829222828149796, "rewards/judge_quality/mean": 0.5387499928474426, "rewards/judge_quality/std": 0.24485784769058228, "rewards/total_composite/mean": 0.6497100591659546, "rewards/total_composite/std": 0.13361245393753052, "reward": 0.6497100591659546, "reward_std": 0.13361245393753052, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12940683960914612, "sampling/sampling_logp_difference/max": 1.7096335887908936, "sampling/importance_sampling_ratio/min": 0.18093208968639374, "sampling/importance_sampling_ratio/mean": 1.009081482887268, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7946780137717724, "clip_ratio/low_mean": 0.08262495556846261, "clip_ratio/low_min": 0.08262495556846261, "clip_ratio/high_mean": 0.03240481577813625, "clip_ratio/high_max": 0.03240481577813625, "clip_ratio/region_mean": 0.11502977134659886, "reward_total_mean": 0.6497100591659546, "reward_meter_mean": 0.9228650331497192, "reward_meter_std": 0.10433200746774673, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.895309329032898, "reward_repeat_soft_std": 0.07829222828149796, "reward_judge_quality_mean": 0.5387499928474426, "reward_judge_quality_std": 0.24485784769058228, "reward_total_composite_mean": 0.6497100591659546, "reward_total_composite_std": 0.13361245393753052} {"timestamp_utc": "2026-04-13T11:55:00Z", "mode": "train", "global_step": 1853, "epoch": 0.18613761928679057, "loss": -0.2299, "grad_norm": 1.8711302280426025, "learning_rate": 4.387878787878788e-06, "num_tokens": 3326861.0, "completions/mean_length": 248.75, "completions/min_length": 192.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 211.1428680419922, "completions/min_terminated_length": 192.0, "completions/max_terminated_length": 244.0, "rewards/meter/mean": 0.9305763244628906, "rewards/meter/std": 0.14726680517196655, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.05892555043101311, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8247259855270386, "rewards/repeat_soft/std": 0.057971034198999405, "rewards/judge_quality/mean": 0.29374998807907104, "rewards/judge_quality/std": 0.1399936079978943, "rewards/total_composite/mean": 0.4310820698738098, "rewards/total_composite/std": 0.1934940665960312, "reward": 0.4310820698738098, "reward_std": 0.19349405169487, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10495075583457947, "sampling/sampling_logp_difference/max": 1.4805593490600586, "sampling/importance_sampling_ratio/min": 0.22751040756702423, "sampling/importance_sampling_ratio/mean": 1.0249090194702148, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8760777711868286, "clip_ratio/low_mean": 0.01909655425697565, "clip_ratio/low_min": 0.01909655425697565, "clip_ratio/high_mean": 0.059721045196056366, "clip_ratio/high_max": 0.059721045196056366, "clip_ratio/region_mean": 0.07881759945303202, "reward_total_mean": 0.4310820698738098, "reward_meter_mean": 0.9305763244628906, "reward_meter_std": 0.14726680517196655, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.05892555043101311, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8247259855270386, "reward_repeat_soft_std": 0.057971034198999405, "reward_judge_quality_mean": 0.29374998807907104, "reward_judge_quality_std": 0.1399936079978943, "reward_total_composite_mean": 0.4310820698738098, "reward_total_composite_std": 0.1934940665960312} {"timestamp_utc": "2026-04-13T11:55:07Z", "mode": "train", "global_step": 1854, "epoch": 0.18623807132094425, "loss": -0.0435, "grad_norm": 6.689228534698486, "learning_rate": 4.384848484848485e-06, "num_tokens": 3328859.0, "completions/mean_length": 83.75, "completions/min_length": 52.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 83.75, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.8792723417282104, "rewards/meter/std": 0.24548029899597168, "rewards/count_adherence/mean": 0.7083333730697632, "rewards/count_adherence/std": 0.11785111576318741, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7887636423110962, "rewards/repeat_soft/std": 0.053252626210451126, "rewards/judge_quality/mean": 0.5824999809265137, "rewards/judge_quality/std": 0.23260943591594696, "rewards/total_composite/mean": 0.5860437750816345, "rewards/total_composite/std": 0.16476160287857056, "reward": 0.5860437750816345, "reward_std": 0.16476160287857056, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08537229150533676, "sampling/sampling_logp_difference/max": 1.5889043807983398, "sampling/importance_sampling_ratio/min": 0.20414915680885315, "sampling/importance_sampling_ratio/mean": 1.0225580930709839, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7083901986479759, "clip_ratio/low_mean": 0.040940255392342806, "clip_ratio/low_min": 0.040940255392342806, "clip_ratio/high_mean": 0.022865363396704197, "clip_ratio/high_max": 0.022865363396704197, "clip_ratio/region_mean": 0.063805618789047, "reward_total_mean": 0.5860437750816345, "reward_meter_mean": 0.8792723417282104, "reward_meter_std": 0.24548029899597168, "reward_count_adherence_mean": 0.7083333730697632, "reward_count_adherence_std": 0.11785111576318741, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7887636423110962, "reward_repeat_soft_std": 0.053252626210451126, "reward_judge_quality_mean": 0.5824999809265137, "reward_judge_quality_std": 0.23260943591594696, "reward_total_composite_mean": 0.5860437750816345, "reward_total_composite_std": 0.16476160287857056} {"timestamp_utc": "2026-04-13T11:55:18Z", "mode": "train", "global_step": 1855, "epoch": 0.18633852335509793, "loss": -0.1029, "grad_norm": 1.8869119882583618, "learning_rate": 4.381818181818182e-06, "num_tokens": 3330471.0, "completions/mean_length": 166.5, "completions/min_length": 40.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 51.333335876464844, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.5134392976760864, "rewards/meter/std": 0.44988754391670227, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.98429274559021, "rewards/repeat_soft/std": 0.0269598551094532, "rewards/judge_quality/mean": 0.4350000023841858, "rewards/judge_quality/std": 0.29611775279045105, "rewards/total_composite/mean": 0.428304523229599, "rewards/total_composite/std": 0.3222726583480835, "reward": 0.428304523229599, "reward_std": 0.3222726285457611, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14015433192253113, "sampling/sampling_logp_difference/max": 1.8146681785583496, "sampling/importance_sampling_ratio/min": 0.1628919541835785, "sampling/importance_sampling_ratio/mean": 1.0194242000579834, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5727350860834122, "clip_ratio/low_mean": 0.03384146373718977, "clip_ratio/low_min": 0.03384146373718977, "clip_ratio/high_mean": 0.06243182811886072, "clip_ratio/high_max": 0.06243182811886072, "clip_ratio/region_mean": 0.09627329185605049, "reward_total_mean": 0.428304523229599, "reward_meter_mean": 0.5134392976760864, "reward_meter_std": 0.44988754391670227, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.98429274559021, "reward_repeat_soft_std": 0.0269598551094532, "reward_judge_quality_mean": 0.4350000023841858, "reward_judge_quality_std": 0.29611775279045105, "reward_total_composite_mean": 0.428304523229599, "reward_total_composite_std": 0.3222726583480835} {"timestamp_utc": "2026-04-13T11:55:25Z", "mode": "train", "global_step": 1856, "epoch": 0.18643897538925164, "loss": -0.0603, "grad_norm": 12.850541114807129, "learning_rate": 4.378787878787879e-06, "num_tokens": 3332075.0, "completions/mean_length": 36.5, "completions/min_length": 27.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.5, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.5566971302032471, "rewards/meter/std": 0.46029070019721985, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9029535055160522, "rewards/repeat_soft/std": 0.053695887327194214, "rewards/judge_quality/mean": 0.8575000166893005, "rewards/judge_quality/std": 0.176776722073555, "rewards/total_composite/mean": 0.6285346150398254, "rewards/total_composite/std": 0.2570481300354004, "reward": 0.6285346150398254, "reward_std": 0.257048100233078, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09701843559741974, "sampling/sampling_logp_difference/max": 1.2670092582702637, "sampling/importance_sampling_ratio/min": 0.28167277574539185, "sampling/importance_sampling_ratio/mean": 1.0066252946853638, "sampling/importance_sampling_ratio/max": 1.831130862236023, "entropy": 0.7173832096159458, "clip_ratio/low_mean": 0.04563816008158028, "clip_ratio/low_min": 0.04563816008158028, "clip_ratio/high_mean": 0.04727564286440611, "clip_ratio/high_max": 0.04727564286440611, "clip_ratio/region_mean": 0.09291380294598639, "reward_total_mean": 0.6285346150398254, "reward_meter_mean": 0.5566971302032471, "reward_meter_std": 0.46029070019721985, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9029535055160522, "reward_repeat_soft_std": 0.053695887327194214, "reward_judge_quality_mean": 0.8575000166893005, "reward_judge_quality_std": 0.176776722073555, "reward_total_composite_mean": 0.6285346150398254, "reward_total_composite_std": 0.2570481300354004} {"timestamp_utc": "2026-04-13T11:55:31Z", "mode": "train", "global_step": 1857, "epoch": 0.18653942742340532, "loss": -0.0377, "grad_norm": 10.985547065734863, "learning_rate": 4.375757575757576e-06, "num_tokens": 3333626.0, "completions/mean_length": 33.875, "completions/min_length": 29.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.875, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.616787314414978, "rewards/meter/std": 0.42414337396621704, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9226333498954773, "rewards/repeat_soft/std": 0.03281796723604202, "rewards/judge_quality/mean": 0.9200000166893005, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7072337865829468, "rewards/total_composite/std": 0.25289255380630493, "reward": 0.7072337865829468, "reward_std": 0.25289252400398254, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08271818608045578, "sampling/sampling_logp_difference/max": 1.549966812133789, "sampling/importance_sampling_ratio/min": 0.21225501596927643, "sampling/importance_sampling_ratio/mean": 0.9723013043403625, "sampling/importance_sampling_ratio/max": 1.7442371845245361, "entropy": 0.27876042760908604, "clip_ratio/low_mean": 0.008342602755874395, "clip_ratio/low_min": 0.008342602755874395, "clip_ratio/high_mean": 0.052414074540138245, "clip_ratio/high_max": 0.052414074540138245, "clip_ratio/region_mean": 0.06075667729601264, "reward_total_mean": 0.7072337865829468, "reward_meter_mean": 0.616787314414978, "reward_meter_std": 0.42414337396621704, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9226333498954773, "reward_repeat_soft_std": 0.03281796723604202, "reward_judge_quality_mean": 0.9200000166893005, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7072337865829468, "reward_total_composite_std": 0.25289255380630493} {"timestamp_utc": "2026-04-13T11:55:39Z", "mode": "train", "global_step": 1858, "epoch": 0.18663987945755903, "loss": -0.0044, "grad_norm": 5.345325946807861, "learning_rate": 4.372727272727273e-06, "num_tokens": 3336330.0, "completions/mean_length": 157.0, "completions/min_length": 137.0, "completions/max_length": 180.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 157.0, "completions/min_terminated_length": 137.0, "completions/max_terminated_length": 180.0, "rewards/meter/mean": 0.8315615057945251, "rewards/meter/std": 0.23402608931064606, "rewards/count_adherence/mean": 0.53125, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8731966018676758, "rewards/repeat_soft/std": 0.04789191856980324, "rewards/judge_quality/mean": 0.4612500071525574, "rewards/judge_quality/std": 0.19467465579509735, "rewards/total_composite/mean": 0.4668791592121124, "rewards/total_composite/std": 0.06063517555594444, "reward": 0.4668791592121124, "reward_std": 0.060635171830654144, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10826198756694794, "sampling/sampling_logp_difference/max": 2.527419090270996, "sampling/importance_sampling_ratio/min": 0.07986488193273544, "sampling/importance_sampling_ratio/mean": 1.015867829322815, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.742377758026123, "clip_ratio/low_mean": 0.04996927920728922, "clip_ratio/low_min": 0.04996927920728922, "clip_ratio/high_mean": 0.05791372526437044, "clip_ratio/high_max": 0.05791372526437044, "clip_ratio/region_mean": 0.10788300447165966, "reward_total_mean": 0.4668791592121124, "reward_meter_mean": 0.8315615057945251, "reward_meter_std": 0.23402608931064606, "reward_count_adherence_mean": 0.53125, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8731966018676758, "reward_repeat_soft_std": 0.04789191856980324, "reward_judge_quality_mean": 0.4612500071525574, "reward_judge_quality_std": 0.19467465579509735, "reward_total_composite_mean": 0.4668791592121124, "reward_total_composite_std": 0.06063517555594444} {"timestamp_utc": "2026-04-13T11:55:47Z", "mode": "train", "global_step": 1859, "epoch": 0.1867403314917127, "loss": 0.0957, "grad_norm": 7.484099388122559, "learning_rate": 4.36969696969697e-06, "num_tokens": 3338346.0, "completions/mean_length": 63.0, "completions/min_length": 52.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.0, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.9657058119773865, "rewards/meter/std": 0.005547062493860722, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.17251639068126678, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7004228830337524, "rewards/repeat_soft/std": 0.06686412543058395, "rewards/judge_quality/mean": 0.29249998927116394, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.46350228786468506, "rewards/total_composite/std": 0.06039566174149513, "reward": 0.46350228786468506, "reward_std": 0.06039566174149513, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0521131195127964, "sampling/sampling_logp_difference/max": 1.4458332061767578, "sampling/importance_sampling_ratio/min": 0.23554974794387817, "sampling/importance_sampling_ratio/mean": 0.9977630972862244, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2415139451622963, "clip_ratio/low_mean": 0.006274366518482566, "clip_ratio/low_min": 0.006274366518482566, "clip_ratio/high_mean": 0.0413695378229022, "clip_ratio/high_max": 0.0413695378229022, "clip_ratio/region_mean": 0.04764390434138477, "reward_total_mean": 0.46350228786468506, "reward_meter_mean": 0.9657058119773865, "reward_meter_std": 0.005547062493860722, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.17251639068126678, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7004228830337524, "reward_repeat_soft_std": 0.06686412543058395, "reward_judge_quality_mean": 0.29249998927116394, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.46350228786468506, "reward_total_composite_std": 0.06039566174149513} {"timestamp_utc": "2026-04-13T11:55:54Z", "mode": "train", "global_step": 1860, "epoch": 0.1868407835258664, "loss": -0.0241, "grad_norm": 5.998904228210449, "learning_rate": 4.366666666666667e-06, "num_tokens": 3340688.0, "completions/mean_length": 108.75, "completions/min_length": 82.0, "completions/max_length": 127.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.75, "completions/min_terminated_length": 82.0, "completions/max_terminated_length": 127.0, "rewards/meter/mean": 0.8752556443214417, "rewards/meter/std": 0.22284598648548126, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7929337024688721, "rewards/repeat_soft/std": 0.03223147615790367, "rewards/judge_quality/mean": 0.4362500011920929, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.5178137421607971, "rewards/total_composite/std": 0.10374221205711365, "reward": 0.5178137421607971, "reward_std": 0.10374220460653305, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08187668025493622, "sampling/sampling_logp_difference/max": 2.881627082824707, "sampling/importance_sampling_ratio/min": 0.05604349821805954, "sampling/importance_sampling_ratio/mean": 1.0006663799285889, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5625111572444439, "clip_ratio/low_mean": 0.036750719882547855, "clip_ratio/low_min": 0.036750719882547855, "clip_ratio/high_mean": 0.0497631561011076, "clip_ratio/high_max": 0.0497631561011076, "clip_ratio/region_mean": 0.08651387598365545, "reward_total_mean": 0.5178137421607971, "reward_meter_mean": 0.8752556443214417, "reward_meter_std": 0.22284598648548126, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7929337024688721, "reward_repeat_soft_std": 0.03223147615790367, "reward_judge_quality_mean": 0.4362500011920929, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.5178137421607971, "reward_total_composite_std": 0.10374221205711365} {"timestamp_utc": "2026-04-13T11:56:02Z", "mode": "train", "global_step": 1861, "epoch": 0.1869412355600201, "loss": 0.0764, "grad_norm": 7.427359580993652, "learning_rate": 4.363636363636364e-06, "num_tokens": 3343149.0, "completions/mean_length": 129.625, "completions/min_length": 110.0, "completions/max_length": 149.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 129.625, "completions/min_terminated_length": 110.0, "completions/max_terminated_length": 149.0, "rewards/meter/mean": 0.9266027212142944, "rewards/meter/std": 0.07922521233558655, "rewards/count_adherence/mean": 0.7749999761581421, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8284742832183838, "rewards/repeat_soft/std": 0.07496636360883713, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.4970102608203888, "rewards/total_composite/std": 0.07223185896873474, "reward": 0.4970102608203888, "reward_std": 0.07223183661699295, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10745149105787277, "sampling/sampling_logp_difference/max": 1.8511078357696533, "sampling/importance_sampling_ratio/min": 0.18094448745250702, "sampling/importance_sampling_ratio/mean": 1.0111578702926636, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6480537839233875, "clip_ratio/low_mean": 0.03165066707879305, "clip_ratio/low_min": 0.03165066707879305, "clip_ratio/high_mean": 0.05648232623934746, "clip_ratio/high_max": 0.05648232623934746, "clip_ratio/region_mean": 0.0881329933181405, "reward_total_mean": 0.4970102608203888, "reward_meter_mean": 0.9266027212142944, "reward_meter_std": 0.07922521233558655, "reward_count_adherence_mean": 0.7749999761581421, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8284742832183838, "reward_repeat_soft_std": 0.07496636360883713, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.4970102608203888, "reward_total_composite_std": 0.07223185896873474} {"timestamp_utc": "2026-04-13T11:56:14Z", "mode": "train", "global_step": 1862, "epoch": 0.18704168759417378, "loss": -0.1356, "grad_norm": 2.285029888153076, "learning_rate": 4.36060606060606e-06, "num_tokens": 3344944.0, "completions/mean_length": 108.375, "completions/min_length": 39.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 50.71428680419922, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.8659183979034424, "rewards/meter/std": 0.2861400246620178, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8929972648620605, "rewards/repeat_soft/std": 0.02832763083279133, "rewards/judge_quality/mean": 0.5649999976158142, "rewards/judge_quality/std": 0.32062438130378723, "rewards/total_composite/mean": 0.6402934193611145, "rewards/total_composite/std": 0.2991735339164734, "reward": 0.6402934193611145, "reward_std": 0.2991735339164734, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10415394604206085, "sampling/sampling_logp_difference/max": 1.1604008674621582, "sampling/importance_sampling_ratio/min": 0.3133605420589447, "sampling/importance_sampling_ratio/mean": 1.0106756687164307, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7009025365114212, "clip_ratio/low_mean": 0.056871891021728516, "clip_ratio/low_min": 0.056871891021728516, "clip_ratio/high_mean": 0.03592592664062977, "clip_ratio/high_max": 0.03592592664062977, "clip_ratio/region_mean": 0.09279781766235828, "reward_total_mean": 0.6402934193611145, "reward_meter_mean": 0.8659183979034424, "reward_meter_std": 0.2861400246620178, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8929972648620605, "reward_repeat_soft_std": 0.02832763083279133, "reward_judge_quality_mean": 0.5649999976158142, "reward_judge_quality_std": 0.32062438130378723, "reward_total_composite_mean": 0.6402934193611145, "reward_total_composite_std": 0.2991735339164734} {"timestamp_utc": "2026-04-13T11:56:23Z", "mode": "train", "global_step": 1863, "epoch": 0.1871421396283275, "loss": 0.0731, "grad_norm": 4.1012701988220215, "learning_rate": 4.3575757575757576e-06, "num_tokens": 3348524.0, "completions/mean_length": 238.5, "completions/min_length": 220.0, "completions/max_length": 260.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 238.5, "completions/min_terminated_length": 220.0, "completions/max_terminated_length": 260.0, "rewards/meter/mean": 0.9505437612533569, "rewards/meter/std": 0.042091019451618195, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.08908706158399582, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6943099498748779, "rewards/repeat_soft/std": 0.03186754137277603, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5136449337005615, "rewards/total_composite/std": 0.02377447858452797, "reward": 0.5136449337005615, "reward_std": 0.023774469271302223, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06255806237459183, "sampling/sampling_logp_difference/max": 1.5884747505187988, "sampling/importance_sampling_ratio/min": 0.20423687994480133, "sampling/importance_sampling_ratio/mean": 1.001752495765686, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.368648037314415, "clip_ratio/low_mean": 0.022642745170742273, "clip_ratio/low_min": 0.022642745170742273, "clip_ratio/high_mean": 0.02703581377863884, "clip_ratio/high_max": 0.02703581377863884, "clip_ratio/region_mean": 0.04967855894938111, "reward_total_mean": 0.5136449337005615, "reward_meter_mean": 0.9505437612533569, "reward_meter_std": 0.042091019451618195, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.08908706158399582, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6943099498748779, "reward_repeat_soft_std": 0.03186754137277603, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5136449337005615, "reward_total_composite_std": 0.02377447858452797} {"timestamp_utc": "2026-04-13T11:56:30Z", "mode": "train", "global_step": 1864, "epoch": 0.18724259166248117, "loss": -0.0471, "grad_norm": 6.634945392608643, "learning_rate": 4.354545454545455e-06, "num_tokens": 3350888.0, "completions/mean_length": 122.5, "completions/min_length": 108.0, "completions/max_length": 139.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.5, "completions/min_terminated_length": 108.0, "completions/max_terminated_length": 139.0, "rewards/meter/mean": 0.8727961182594299, "rewards/meter/std": 0.24367918074131012, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9506475329399109, "rewards/repeat_soft/std": 0.019312597811222076, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.5545367002487183, "rewards/total_composite/std": 0.14379402995109558, "reward": 0.5545367002487183, "reward_std": 0.1437940150499344, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12057143449783325, "sampling/sampling_logp_difference/max": 1.7719669342041016, "sampling/importance_sampling_ratio/min": 0.16999828815460205, "sampling/importance_sampling_ratio/mean": 1.0171279907226562, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9396882504224777, "clip_ratio/low_mean": 0.08226289879530668, "clip_ratio/low_min": 0.08226289879530668, "clip_ratio/high_mean": 0.008992806077003479, "clip_ratio/high_max": 0.008992806077003479, "clip_ratio/region_mean": 0.09125570487231016, "reward_total_mean": 0.5545367002487183, "reward_meter_mean": 0.8727961182594299, "reward_meter_std": 0.24367918074131012, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9506475329399109, "reward_repeat_soft_std": 0.019312597811222076, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.5545367002487183, "reward_total_composite_std": 0.14379402995109558} {"timestamp_utc": "2026-04-13T11:56:37Z", "mode": "train", "global_step": 1865, "epoch": 0.18734304369663485, "loss": 0.1142, "grad_norm": 13.94418716430664, "learning_rate": 4.351515151515152e-06, "num_tokens": 3352545.0, "completions/mean_length": 34.125, "completions/min_length": 27.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.125, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.5891563892364502, "rewards/meter/std": 0.42681676149368286, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9475993514060974, "rewards/repeat_soft/std": 0.04067608714103699, "rewards/judge_quality/mean": 0.7950000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.6522279977798462, "rewards/total_composite/std": 0.23976483941078186, "reward": 0.6522279977798462, "reward_std": 0.23976485431194305, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10949678719043732, "sampling/sampling_logp_difference/max": 1.8282232284545898, "sampling/importance_sampling_ratio/min": 0.16069884598255157, "sampling/importance_sampling_ratio/mean": 1.0080785751342773, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5568082258105278, "clip_ratio/low_mean": 0.052935092244297266, "clip_ratio/low_min": 0.052935092244297266, "clip_ratio/high_mean": 0.03388047218322754, "clip_ratio/high_max": 0.03388047218322754, "clip_ratio/region_mean": 0.0868155644275248, "reward_total_mean": 0.6522279977798462, "reward_meter_mean": 0.5891563892364502, "reward_meter_std": 0.42681676149368286, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9475993514060974, "reward_repeat_soft_std": 0.04067608714103699, "reward_judge_quality_mean": 0.7950000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.6522279977798462, "reward_total_composite_std": 0.23976483941078186} {"timestamp_utc": "2026-04-13T11:56:45Z", "mode": "train", "global_step": 1866, "epoch": 0.18744349573078856, "loss": -0.0036, "grad_norm": 6.008486747741699, "learning_rate": 4.348484848484849e-06, "num_tokens": 3355173.0, "completions/mean_length": 154.5, "completions/min_length": 143.0, "completions/max_length": 172.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 154.5, "completions/min_terminated_length": 143.0, "completions/max_terminated_length": 172.0, "rewards/meter/mean": 0.8012826442718506, "rewards/meter/std": 0.2174297273159027, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8598213195800781, "rewards/repeat_soft/std": 0.027137139812111855, "rewards/judge_quality/mean": 0.5824999809265137, "rewards/judge_quality/std": 0.23260943591594696, "rewards/total_composite/mean": 0.5909258723258972, "rewards/total_composite/std": 0.16364625096321106, "reward": 0.5909258723258972, "reward_std": 0.16364626586437225, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08679886162281036, "sampling/sampling_logp_difference/max": 1.854860782623291, "sampling/importance_sampling_ratio/min": 0.1564747393131256, "sampling/importance_sampling_ratio/mean": 1.002786636352539, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5204651392996311, "clip_ratio/low_mean": 0.04906986514106393, "clip_ratio/low_min": 0.04906986514106393, "clip_ratio/high_mean": 0.033117182552814484, "clip_ratio/high_max": 0.033117182552814484, "clip_ratio/region_mean": 0.08218704769387841, "reward_total_mean": 0.5909258723258972, "reward_meter_mean": 0.8012826442718506, "reward_meter_std": 0.2174297273159027, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8598213195800781, "reward_repeat_soft_std": 0.027137139812111855, "reward_judge_quality_mean": 0.5824999809265137, "reward_judge_quality_std": 0.23260943591594696, "reward_total_composite_mean": 0.5909258723258972, "reward_total_composite_std": 0.16364625096321106} {"timestamp_utc": "2026-04-13T11:56:51Z", "mode": "train", "global_step": 1867, "epoch": 0.18754394776494224, "loss": -0.0284, "grad_norm": 11.911911964416504, "learning_rate": 4.345454545454546e-06, "num_tokens": 3356930.0, "completions/mean_length": 39.625, "completions/min_length": 29.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.625, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.6499167680740356, "rewards/meter/std": 0.30164313316345215, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9578688144683838, "rewards/repeat_soft/std": 0.04335197061300278, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.52110755443573, "rewards/total_composite/std": 0.08668576180934906, "reward": 0.52110755443573, "reward_std": 0.08668576180934906, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10192485898733139, "sampling/sampling_logp_difference/max": 0.9830365180969238, "sampling/importance_sampling_ratio/min": 0.37417319416999817, "sampling/importance_sampling_ratio/mean": 1.0202760696411133, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6632933504879475, "clip_ratio/low_mean": 0.04292092937976122, "clip_ratio/low_min": 0.04292092937976122, "clip_ratio/high_mean": 0.05193390650674701, "clip_ratio/high_max": 0.05193390650674701, "clip_ratio/region_mean": 0.09485483588650823, "reward_total_mean": 0.52110755443573, "reward_meter_mean": 0.6499167680740356, "reward_meter_std": 0.30164313316345215, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9578688144683838, "reward_repeat_soft_std": 0.04335197061300278, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.52110755443573, "reward_total_composite_std": 0.08668576180934906} {"timestamp_utc": "2026-04-13T11:56:58Z", "mode": "train", "global_step": 1868, "epoch": 0.18764439979909592, "loss": 0.0075, "grad_norm": 10.356154441833496, "learning_rate": 4.342424242424243e-06, "num_tokens": 3358867.0, "completions/mean_length": 61.125, "completions/min_length": 56.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.125, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.7613774538040161, "rewards/meter/std": 0.31797435879707336, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.995015025138855, "rewards/repeat_soft/std": 0.007305712439119816, "rewards/judge_quality/mean": 0.7362500429153442, "rewards/judge_quality/std": 0.25376805663108826, "rewards/total_composite/mean": 0.6909343004226685, "rewards/total_composite/std": 0.1758246123790741, "reward": 0.6909343004226685, "reward_std": 0.1758246123790741, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11673297733068466, "sampling/sampling_logp_difference/max": 1.9016571044921875, "sampling/importance_sampling_ratio/min": 0.14932097494602203, "sampling/importance_sampling_ratio/mean": 0.991946280002594, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7148168757557869, "clip_ratio/low_mean": 0.05922755692154169, "clip_ratio/low_min": 0.05922755692154169, "clip_ratio/high_mean": 0.046878415159881115, "clip_ratio/high_max": 0.046878415159881115, "clip_ratio/region_mean": 0.1061059720814228, "reward_total_mean": 0.6909343004226685, "reward_meter_mean": 0.7613774538040161, "reward_meter_std": 0.31797435879707336, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.995015025138855, "reward_repeat_soft_std": 0.007305712439119816, "reward_judge_quality_mean": 0.7362500429153442, "reward_judge_quality_std": 0.25376805663108826, "reward_total_composite_mean": 0.6909343004226685, "reward_total_composite_std": 0.1758246123790741} {"timestamp_utc": "2026-04-13T11:57:04Z", "mode": "train", "global_step": 1869, "epoch": 0.18774485183324963, "loss": -0.0939, "grad_norm": 21.15143585205078, "learning_rate": 4.33939393939394e-06, "num_tokens": 3360190.0, "completions/mean_length": 21.375, "completions/min_length": 17.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.375, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.8383228778839111, "rewards/meter/std": 0.27383944392204285, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9510330557823181, "rewards/repeat_soft/std": 0.027066897600889206, "rewards/judge_quality/mean": 0.4437500238418579, "rewards/judge_quality/std": 0.2084594964981079, "rewards/total_composite/mean": 0.5913371443748474, "rewards/total_composite/std": 0.15817147493362427, "reward": 0.5913371443748474, "reward_std": 0.15817148983478546, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13284669816493988, "sampling/sampling_logp_difference/max": 1.202202320098877, "sampling/importance_sampling_ratio/min": 0.30053162574768066, "sampling/importance_sampling_ratio/mean": 1.0145890712738037, "sampling/importance_sampling_ratio/max": 1.8619064092636108, "entropy": 1.180416114628315, "clip_ratio/low_mean": 0.03137254947796464, "clip_ratio/low_min": 0.03137254947796464, "clip_ratio/high_mean": 0.06869431119412184, "clip_ratio/high_max": 0.06869431119412184, "clip_ratio/region_mean": 0.10006686067208648, "reward_total_mean": 0.5913371443748474, "reward_meter_mean": 0.8383228778839111, "reward_meter_std": 0.27383944392204285, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9510330557823181, "reward_repeat_soft_std": 0.027066897600889206, "reward_judge_quality_mean": 0.4437500238418579, "reward_judge_quality_std": 0.2084594964981079, "reward_total_composite_mean": 0.5913371443748474, "reward_total_composite_std": 0.15817147493362427} {"timestamp_utc": "2026-04-13T11:57:10Z", "mode": "train", "global_step": 1870, "epoch": 0.1878453038674033, "loss": 0.0627, "grad_norm": 15.947404861450195, "learning_rate": 4.336363636363637e-06, "num_tokens": 3361583.0, "completions/mean_length": 32.125, "completions/min_length": 24.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.125, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.5034498572349548, "rewards/meter/std": 0.43890899419784546, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.938712477684021, "rewards/repeat_soft/std": 0.04584834352135658, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.478248655796051, "rewards/total_composite/std": 0.11615147441625595, "reward": 0.478248655796051, "reward_std": 0.11615145951509476, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08563661575317383, "sampling/sampling_logp_difference/max": 1.4977467060089111, "sampling/importance_sampling_ratio/min": 0.2236335128545761, "sampling/importance_sampling_ratio/mean": 0.9934297204017639, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4042210765182972, "clip_ratio/low_mean": 0.02282913215458393, "clip_ratio/low_min": 0.02282913215458393, "clip_ratio/high_mean": 0.04942481033504009, "clip_ratio/high_max": 0.04942481033504009, "clip_ratio/region_mean": 0.07225394248962402, "reward_total_mean": 0.478248655796051, "reward_meter_mean": 0.5034498572349548, "reward_meter_std": 0.43890899419784546, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.938712477684021, "reward_repeat_soft_std": 0.04584834352135658, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.478248655796051, "reward_total_composite_std": 0.11615147441625595} {"timestamp_utc": "2026-04-13T11:57:16Z", "mode": "train", "global_step": 1871, "epoch": 0.18794575590155702, "loss": 0.036, "grad_norm": 8.205052375793457, "learning_rate": 4.333333333333334e-06, "num_tokens": 3363452.0, "completions/mean_length": 60.625, "completions/min_length": 51.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.625, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.926069974899292, "rewards/meter/std": 0.14992724359035492, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9523845911026001, "rewards/repeat_soft/std": 0.03608405217528343, "rewards/judge_quality/mean": 0.9200000166893005, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8966474533081055, "rewards/total_composite/std": 0.09090722352266312, "reward": 0.8966474533081055, "reward_std": 0.09090722352266312, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11477474868297577, "sampling/sampling_logp_difference/max": 3.223674774169922, "sampling/importance_sampling_ratio/min": 0.03980850428342819, "sampling/importance_sampling_ratio/mean": 1.0129293203353882, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6815303079783916, "clip_ratio/low_mean": 0.015384615398943424, "clip_ratio/low_min": 0.015384615398943424, "clip_ratio/high_mean": 0.1016759155318141, "clip_ratio/high_max": 0.1016759155318141, "clip_ratio/region_mean": 0.11706053093075752, "reward_total_mean": 0.8966474533081055, "reward_meter_mean": 0.926069974899292, "reward_meter_std": 0.14992724359035492, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9523845911026001, "reward_repeat_soft_std": 0.03608405217528343, "reward_judge_quality_mean": 0.9200000166893005, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8966474533081055, "reward_total_composite_std": 0.09090722352266312} {"timestamp_utc": "2026-04-13T11:57:33Z", "mode": "train", "global_step": 1872, "epoch": 0.1880462079357107, "loss": -0.0684, "grad_norm": 2.9684641361236572, "learning_rate": 4.330303030303031e-06, "num_tokens": 3365012.0, "completions/mean_length": 89.0, "completions/min_length": 25.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 28.571430206298828, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.7443959712982178, "rewards/meter/std": 0.4385094940662384, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9671874642372131, "rewards/repeat_soft/std": 0.013258260674774647, "rewards/judge_quality/mean": 0.5024999976158142, "rewards/judge_quality/std": 0.2886792719364166, "rewards/total_composite/mean": 0.5870416164398193, "rewards/total_composite/std": 0.3021523356437683, "reward": 0.5870416164398193, "reward_std": 0.3021523058414459, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10582827031612396, "sampling/sampling_logp_difference/max": 0.9363492727279663, "sampling/importance_sampling_ratio/min": 0.39205652475357056, "sampling/importance_sampling_ratio/mean": 0.9953686594963074, "sampling/importance_sampling_ratio/max": 1.7579275369644165, "entropy": 0.5372766740620136, "clip_ratio/low_mean": 0.02016128972172737, "clip_ratio/low_min": 0.02016128972172737, "clip_ratio/high_mean": 0.08254634728655219, "clip_ratio/high_max": 0.08254634728655219, "clip_ratio/region_mean": 0.10270763700827956, "reward_total_mean": 0.5870416164398193, "reward_meter_mean": 0.7443959712982178, "reward_meter_std": 0.4385094940662384, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9671874642372131, "reward_repeat_soft_std": 0.013258260674774647, "reward_judge_quality_mean": 0.5024999976158142, "reward_judge_quality_std": 0.2886792719364166, "reward_total_composite_mean": 0.5870416164398193, "reward_total_composite_std": 0.3021523356437683} {"timestamp_utc": "2026-04-13T11:57:44Z", "mode": "train", "global_step": 1873, "epoch": 0.18814665996986438, "loss": -0.1429, "grad_norm": 1.6155564785003662, "learning_rate": 4.327272727272728e-06, "num_tokens": 3366714.0, "completions/mean_length": 109.75, "completions/min_length": 46.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 52.28571701049805, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9725117683410645, "rewards/meter/std": 0.025253908708691597, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.888059139251709, "rewards/repeat_soft/std": 0.11275843530893326, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.13617216050624847, "rewards/total_composite/mean": 0.530485212802887, "rewards/total_composite/std": 0.21530216932296753, "reward": 0.530485212802887, "reward_std": 0.21530215442180634, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09288246929645538, "sampling/sampling_logp_difference/max": 0.9357633590698242, "sampling/importance_sampling_ratio/min": 0.3922863006591797, "sampling/importance_sampling_ratio/mean": 1.0329862833023071, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5780495665967464, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.06867086002603173, "clip_ratio/high_max": 0.06867086002603173, "clip_ratio/region_mean": 0.06867086002603173, "reward_total_mean": 0.530485212802887, "reward_meter_mean": 0.9725117683410645, "reward_meter_std": 0.025253908708691597, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.888059139251709, "reward_repeat_soft_std": 0.11275843530893326, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.13617216050624847, "reward_total_composite_mean": 0.530485212802887, "reward_total_composite_std": 0.21530216932296753} {"timestamp_utc": "2026-04-13T11:57:51Z", "mode": "train", "global_step": 1874, "epoch": 0.18824711200401809, "loss": 0.078, "grad_norm": 5.653331279754639, "learning_rate": 4.324242424242425e-06, "num_tokens": 3368907.0, "completions/mean_length": 99.125, "completions/min_length": 63.0, "completions/max_length": 138.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 99.125, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 138.0, "rewards/meter/mean": 0.6673324108123779, "rewards/meter/std": 0.34372588992118835, "rewards/count_adherence/mean": 0.84375, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.806655764579773, "rewards/repeat_soft/std": 0.03883816674351692, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.5225415825843811, "rewards/total_composite/std": 0.16865621507167816, "reward": 0.5225415825843811, "reward_std": 0.16865621507167816, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08662410825490952, "sampling/sampling_logp_difference/max": 2.349956750869751, "sampling/importance_sampling_ratio/min": 0.09537328779697418, "sampling/importance_sampling_ratio/mean": 1.0094082355499268, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4720325320959091, "clip_ratio/low_mean": 0.03813416324555874, "clip_ratio/low_min": 0.03813416324555874, "clip_ratio/high_mean": 0.043572402093559504, "clip_ratio/high_max": 0.043572402093559504, "clip_ratio/region_mean": 0.08170656533911824, "reward_total_mean": 0.5225415825843811, "reward_meter_mean": 0.6673324108123779, "reward_meter_std": 0.34372588992118835, "reward_count_adherence_mean": 0.84375, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.806655764579773, "reward_repeat_soft_std": 0.03883816674351692, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.5225415825843811, "reward_total_composite_std": 0.16865621507167816} {"timestamp_utc": "2026-04-13T11:57:57Z", "mode": "train", "global_step": 1875, "epoch": 0.18834756403817177, "loss": 0.0816, "grad_norm": 16.872835159301758, "learning_rate": 4.321212121212121e-06, "num_tokens": 3370271.0, "completions/mean_length": 20.5, "completions/min_length": 14.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 20.5, "completions/min_terminated_length": 14.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.760849118232727, "rewards/meter/std": 0.3263269066810608, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9567934274673462, "rewards/repeat_soft/std": 0.016140474006533623, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.555090606212616, "rewards/total_composite/std": 0.09060538560152054, "reward": 0.555090606212616, "reward_std": 0.09060538560152054, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12364717572927475, "sampling/sampling_logp_difference/max": 0.842437744140625, "sampling/importance_sampling_ratio/min": 0.4306594133377075, "sampling/importance_sampling_ratio/mean": 1.0259215831756592, "sampling/importance_sampling_ratio/max": 1.9881998300552368, "entropy": 0.7779849320650101, "clip_ratio/low_mean": 0.05576923117041588, "clip_ratio/low_min": 0.05576923117041588, "clip_ratio/high_mean": 0.06890453398227692, "clip_ratio/high_max": 0.06890453398227692, "clip_ratio/region_mean": 0.1246737651526928, "reward_total_mean": 0.555090606212616, "reward_meter_mean": 0.760849118232727, "reward_meter_std": 0.3263269066810608, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9567934274673462, "reward_repeat_soft_std": 0.016140474006533623, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.555090606212616, "reward_total_composite_std": 0.09060538560152054} {"timestamp_utc": "2026-04-13T11:58:05Z", "mode": "train", "global_step": 1876, "epoch": 0.18844801607232547, "loss": 0.0623, "grad_norm": 6.934006690979004, "learning_rate": 4.3181818181818185e-06, "num_tokens": 3372924.0, "completions/mean_length": 149.625, "completions/min_length": 129.0, "completions/max_length": 186.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 149.625, "completions/min_terminated_length": 129.0, "completions/max_terminated_length": 186.0, "rewards/meter/mean": 0.9845083951950073, "rewards/meter/std": 0.008187457919120789, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8939747214317322, "rewards/repeat_soft/std": 0.06038915738463402, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.5255609750747681, "rewards/total_composite/std": 0.05376099422574043, "reward": 0.5255609750747681, "reward_std": 0.053760990500450134, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1152753233909607, "sampling/sampling_logp_difference/max": 5.3994598388671875, "sampling/importance_sampling_ratio/min": 0.004519021604210138, "sampling/importance_sampling_ratio/mean": 1.0056995153427124, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7588530033826828, "clip_ratio/low_mean": 0.020154332742094994, "clip_ratio/low_min": 0.020154332742094994, "clip_ratio/high_mean": 0.09387489315122366, "clip_ratio/high_max": 0.09387489315122366, "clip_ratio/region_mean": 0.11402922589331865, "reward_total_mean": 0.5255609750747681, "reward_meter_mean": 0.9845083951950073, "reward_meter_std": 0.008187457919120789, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8939747214317322, "reward_repeat_soft_std": 0.06038915738463402, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.5255609750747681, "reward_total_composite_std": 0.05376099422574043} {"timestamp_utc": "2026-04-13T11:58:12Z", "mode": "train", "global_step": 1877, "epoch": 0.18854846810647916, "loss": 0.0987, "grad_norm": 7.469276428222656, "learning_rate": 4.315151515151516e-06, "num_tokens": 3375096.0, "completions/mean_length": 102.5, "completions/min_length": 91.0, "completions/max_length": 122.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.5, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 122.0, "rewards/meter/mean": 0.992778480052948, "rewards/meter/std": 0.0032708081416785717, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8459296226501465, "rewards/repeat_soft/std": 0.05445515364408493, "rewards/judge_quality/mean": 0.32624998688697815, "rewards/judge_quality/std": 0.11350739002227783, "rewards/total_composite/mean": 0.4707384705543518, "rewards/total_composite/std": 0.07509089261293411, "reward": 0.4707384705543518, "reward_std": 0.07509089261293411, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09624933451414108, "sampling/sampling_logp_difference/max": 1.7367115020751953, "sampling/importance_sampling_ratio/min": 0.1760985553264618, "sampling/importance_sampling_ratio/mean": 1.0022552013397217, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.560472272336483, "clip_ratio/low_mean": 0.03613204089924693, "clip_ratio/low_min": 0.03613204089924693, "clip_ratio/high_mean": 0.04738752171397209, "clip_ratio/high_max": 0.04738752171397209, "clip_ratio/region_mean": 0.08351956261321902, "reward_total_mean": 0.4707384705543518, "reward_meter_mean": 0.992778480052948, "reward_meter_std": 0.0032708081416785717, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8459296226501465, "reward_repeat_soft_std": 0.05445515364408493, "reward_judge_quality_mean": 0.32624998688697815, "reward_judge_quality_std": 0.11350739002227783, "reward_total_composite_mean": 0.4707384705543518, "reward_total_composite_std": 0.07509089261293411} {"timestamp_utc": "2026-04-13T11:58:19Z", "mode": "train", "global_step": 1878, "epoch": 0.18864892014063284, "loss": -0.0117, "grad_norm": 6.692818641662598, "learning_rate": 4.312121212121212e-06, "num_tokens": 3376917.0, "completions/mean_length": 59.625, "completions/min_length": 54.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.625, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.893966794013977, "rewards/meter/std": 0.19910527765750885, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7298498153686523, "rewards/repeat_soft/std": 0.05829789489507675, "rewards/judge_quality/mean": 0.36000001430511475, "rewards/judge_quality/std": 0.1302744299173355, "rewards/total_composite/mean": 0.5063621997833252, "rewards/total_composite/std": 0.08025981485843658, "reward": 0.5063621997833252, "reward_std": 0.08025981485843658, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05902862176299095, "sampling/sampling_logp_difference/max": 1.0365785360336304, "sampling/importance_sampling_ratio/min": 0.3546660840511322, "sampling/importance_sampling_ratio/mean": 1.0158969163894653, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.30942083708941936, "clip_ratio/low_mean": 0.027048735413700342, "clip_ratio/low_min": 0.027048735413700342, "clip_ratio/high_mean": 0.031222944846376777, "clip_ratio/high_max": 0.031222944846376777, "clip_ratio/region_mean": 0.05827168026007712, "reward_total_mean": 0.5063621997833252, "reward_meter_mean": 0.893966794013977, "reward_meter_std": 0.19910527765750885, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7298498153686523, "reward_repeat_soft_std": 0.05829789489507675, "reward_judge_quality_mean": 0.36000001430511475, "reward_judge_quality_std": 0.1302744299173355, "reward_total_composite_mean": 0.5063621997833252, "reward_total_composite_std": 0.08025981485843658} {"timestamp_utc": "2026-04-13T11:58:26Z", "mode": "train", "global_step": 1879, "epoch": 0.18874937217478654, "loss": 0.0134, "grad_norm": 6.908302307128906, "learning_rate": 4.309090909090909e-06, "num_tokens": 3379144.0, "completions/mean_length": 99.375, "completions/min_length": 69.0, "completions/max_length": 117.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 99.375, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.9821760058403015, "rewards/meter/std": 0.0065070525743067265, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.1035098284482956, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7847424745559692, "rewards/repeat_soft/std": 0.08276831358671188, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.5845522880554199, "rewards/total_composite/std": 0.06448714435100555, "reward": 0.5845522880554199, "reward_std": 0.06448714435100555, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09772232174873352, "sampling/sampling_logp_difference/max": 1.08424711227417, "sampling/importance_sampling_ratio/min": 0.3381562829017639, "sampling/importance_sampling_ratio/mean": 1.0168607234954834, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5596269220113754, "clip_ratio/low_mean": 0.077846416272223, "clip_ratio/low_min": 0.077846416272223, "clip_ratio/high_mean": 0.010714286006987095, "clip_ratio/high_max": 0.010714286006987095, "clip_ratio/region_mean": 0.08856070227921009, "reward_total_mean": 0.5845522880554199, "reward_meter_mean": 0.9821760058403015, "reward_meter_std": 0.0065070525743067265, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.1035098284482956, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7847424745559692, "reward_repeat_soft_std": 0.08276831358671188, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.5845522880554199, "reward_total_composite_std": 0.06448714435100555} {"timestamp_utc": "2026-04-13T11:58:33Z", "mode": "train", "global_step": 1880, "epoch": 0.18884982420894023, "loss": 0.2086, "grad_norm": 14.22120189666748, "learning_rate": 4.306060606060607e-06, "num_tokens": 3380772.0, "completions/mean_length": 43.5, "completions/min_length": 33.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.5, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.8323148488998413, "rewards/meter/std": 0.3391997814178467, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8871290683746338, "rewards/repeat_soft/std": 0.05498763918876648, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.5602912902832031, "rewards/total_composite/std": 0.08942002058029175, "reward": 0.5602912902832031, "reward_std": 0.08942001312971115, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11437718570232391, "sampling/sampling_logp_difference/max": 1.2057747840881348, "sampling/importance_sampling_ratio/min": 0.29945987462997437, "sampling/importance_sampling_ratio/mean": 1.0102508068084717, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6011449582874775, "clip_ratio/low_mean": 0.017480531707406044, "clip_ratio/low_min": 0.017480531707406044, "clip_ratio/high_mean": 0.08124872390180826, "clip_ratio/high_max": 0.08124872390180826, "clip_ratio/region_mean": 0.0987292556092143, "reward_total_mean": 0.5602912902832031, "reward_meter_mean": 0.8323148488998413, "reward_meter_std": 0.3391997814178467, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8871290683746338, "reward_repeat_soft_std": 0.05498763918876648, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.5602912902832031, "reward_total_composite_std": 0.08942002058029175} {"timestamp_utc": "2026-04-13T11:58:44Z", "mode": "train", "global_step": 1881, "epoch": 0.18895027624309393, "loss": -0.1174, "grad_norm": 2.324276924133301, "learning_rate": 4.303030303030303e-06, "num_tokens": 3382480.0, "completions/mean_length": 106.5, "completions/min_length": 41.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 48.57143020629883, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.6712371110916138, "rewards/meter/std": 0.39124536514282227, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9916514158248901, "rewards/repeat_soft/std": 0.009813602082431316, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.47794660925865173, "rewards/total_composite/std": 0.21686643362045288, "reward": 0.47794660925865173, "reward_std": 0.21686643362045288, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10020927339792252, "sampling/sampling_logp_difference/max": 2.10355281829834, "sampling/importance_sampling_ratio/min": 0.1220221295952797, "sampling/importance_sampling_ratio/mean": 1.0279405117034912, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5319827161729336, "clip_ratio/low_mean": 0.021701388992369175, "clip_ratio/low_min": 0.021701388992369175, "clip_ratio/high_mean": 0.04760117130354047, "clip_ratio/high_max": 0.04760117130354047, "clip_ratio/region_mean": 0.06930256029590964, "reward_total_mean": 0.47794660925865173, "reward_meter_mean": 0.6712371110916138, "reward_meter_std": 0.39124536514282227, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9916514158248901, "reward_repeat_soft_std": 0.009813602082431316, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.47794660925865173, "reward_total_composite_std": 0.21686643362045288} {"timestamp_utc": "2026-04-13T11:58:50Z", "mode": "train", "global_step": 1882, "epoch": 0.18905072827724761, "loss": 0.0771, "grad_norm": 12.32876205444336, "learning_rate": 4.3e-06, "num_tokens": 3384013.0, "completions/mean_length": 28.625, "completions/min_length": 22.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.625, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9835982322692871, "rewards/meter/std": 0.014550337567925453, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9614755511283875, "rewards/repeat_soft/std": 0.002003318862989545, "rewards/judge_quality/mean": 0.39750000834465027, "rewards/judge_quality/std": 0.10110107809305191, "rewards/total_composite/mean": 0.5981491804122925, "rewards/total_composite/std": 0.06445374339818954, "reward": 0.5981491804122925, "reward_std": 0.06445373594760895, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10443581640720367, "sampling/sampling_logp_difference/max": 1.2602341175079346, "sampling/importance_sampling_ratio/min": 0.28358760476112366, "sampling/importance_sampling_ratio/mean": 1.012134075164795, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6408927664160728, "clip_ratio/low_mean": 0.007575757801532745, "clip_ratio/low_min": 0.007575757801532745, "clip_ratio/high_mean": 0.12095134239643812, "clip_ratio/high_max": 0.12095134239643812, "clip_ratio/region_mean": 0.12852710019797087, "reward_total_mean": 0.5981491804122925, "reward_meter_mean": 0.9835982322692871, "reward_meter_std": 0.014550337567925453, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9614755511283875, "reward_repeat_soft_std": 0.002003318862989545, "reward_judge_quality_mean": 0.39750000834465027, "reward_judge_quality_std": 0.10110107809305191, "reward_total_composite_mean": 0.5981491804122925, "reward_total_composite_std": 0.06445374339818954} {"timestamp_utc": "2026-04-13T11:59:02Z", "mode": "train", "global_step": 1883, "epoch": 0.1891511803114013, "loss": -0.212, "grad_norm": 1.3718756437301636, "learning_rate": 4.296969696969698e-06, "num_tokens": 3386258.0, "completions/mean_length": 176.625, "completions/min_length": 97.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 128.71429443359375, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 161.0, "rewards/meter/mean": 0.9235599040985107, "rewards/meter/std": 0.10048312693834305, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.24775780737400055, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.7810044884681702, "rewards/repeat_soft/std": 0.10195527225732803, "rewards/judge_quality/mean": 0.398749977350235, "rewards/judge_quality/std": 0.15733835101127625, "rewards/total_composite/mean": 0.4790325462818146, "rewards/total_composite/std": 0.1957475244998932, "reward": 0.4790325462818146, "reward_std": 0.195747509598732, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0761503279209137, "sampling/sampling_logp_difference/max": 2.0421500205993652, "sampling/importance_sampling_ratio/min": 0.12974944710731506, "sampling/importance_sampling_ratio/mean": 1.0136713981628418, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.39990518242120743, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.061085117515176535, "clip_ratio/high_max": 0.061085117515176535, "clip_ratio/region_mean": 0.061085117515176535, "reward_total_mean": 0.4790325462818146, "reward_meter_mean": 0.9235599040985107, "reward_meter_std": 0.10048312693834305, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.24775780737400055, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.7810044884681702, "reward_repeat_soft_std": 0.10195527225732803, "reward_judge_quality_mean": 0.398749977350235, "reward_judge_quality_std": 0.15733835101127625, "reward_total_composite_mean": 0.4790325462818146, "reward_total_composite_std": 0.1957475244998932} {"timestamp_utc": "2026-04-13T11:59:08Z", "mode": "train", "global_step": 1884, "epoch": 0.189251632345555, "loss": 0.0481, "grad_norm": 12.55371379852295, "learning_rate": 4.293939393939394e-06, "num_tokens": 3387991.0, "completions/mean_length": 42.625, "completions/min_length": 33.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.625, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.9154489636421204, "rewards/meter/std": 0.09620413929224014, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8638527989387512, "rewards/repeat_soft/std": 0.03854406252503395, "rewards/judge_quality/mean": 0.9200000166893005, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.8770164251327515, "rewards/total_composite/std": 0.054819393903017044, "reward": 0.8770164251327515, "reward_std": 0.05481937527656555, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08091326057910919, "sampling/sampling_logp_difference/max": 2.1089088916778564, "sampling/importance_sampling_ratio/min": 0.12137032300233841, "sampling/importance_sampling_ratio/mean": 1.006975769996643, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.43506034091115, "clip_ratio/low_mean": 0.01957590552046895, "clip_ratio/low_min": 0.01957590552046895, "clip_ratio/high_mean": 0.0516329575330019, "clip_ratio/high_max": 0.0516329575330019, "clip_ratio/region_mean": 0.07120886305347085, "reward_total_mean": 0.8770164251327515, "reward_meter_mean": 0.9154489636421204, "reward_meter_std": 0.09620413929224014, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8638527989387512, "reward_repeat_soft_std": 0.03854406252503395, "reward_judge_quality_mean": 0.9200000166893005, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.8770164251327515, "reward_total_composite_std": 0.054819393903017044} {"timestamp_utc": "2026-04-13T11:59:14Z", "mode": "train", "global_step": 1885, "epoch": 0.18935208437970869, "loss": -0.0067, "grad_norm": 17.774818420410156, "learning_rate": 4.290909090909091e-06, "num_tokens": 3389345.0, "completions/mean_length": 20.25, "completions/min_length": 16.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 20.25, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.8867471218109131, "rewards/meter/std": 0.17445634305477142, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9166569113731384, "rewards/repeat_soft/std": 0.10989454388618469, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.22403763234615326, "rewards/total_composite/mean": 0.5397394299507141, "rewards/total_composite/std": 0.26202690601348877, "reward": 0.5397394299507141, "reward_std": 0.26202690601348877, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19745789468288422, "sampling/sampling_logp_difference/max": 1.2882919311523438, "sampling/importance_sampling_ratio/min": 0.2757413685321808, "sampling/importance_sampling_ratio/mean": 1.0029056072235107, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4838181287050247, "clip_ratio/low_mean": 0.07601686753332615, "clip_ratio/low_min": 0.07601686753332615, "clip_ratio/high_mean": 0.1273423619568348, "clip_ratio/high_max": 0.1273423619568348, "clip_ratio/region_mean": 0.20335922949016094, "reward_total_mean": 0.5397394299507141, "reward_meter_mean": 0.8867471218109131, "reward_meter_std": 0.17445634305477142, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9166569113731384, "reward_repeat_soft_std": 0.10989454388618469, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.22403763234615326, "reward_total_composite_mean": 0.5397394299507141, "reward_total_composite_std": 0.26202690601348877} {"timestamp_utc": "2026-04-13T11:59:20Z", "mode": "train", "global_step": 1886, "epoch": 0.1894525364138624, "loss": -0.0556, "grad_norm": 10.604535102844238, "learning_rate": 4.287878787878788e-06, "num_tokens": 3390990.0, "completions/mean_length": 35.625, "completions/min_length": 30.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.625, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.9604357481002808, "rewards/meter/std": 0.0025081527419388294, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8992391228675842, "rewards/repeat_soft/std": 0.05777277797460556, "rewards/judge_quality/mean": 0.4312499761581421, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.6040985584259033, "rewards/total_composite/std": 0.014271872118115425, "reward": 0.6040985584259033, "reward_std": 0.014271865598857403, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06473870575428009, "sampling/sampling_logp_difference/max": 1.0003001689910889, "sampling/importance_sampling_ratio/min": 0.3677690625190735, "sampling/importance_sampling_ratio/mean": 1.0079113245010376, "sampling/importance_sampling_ratio/max": 1.7900887727737427, "entropy": 0.3560235071927309, "clip_ratio/low_mean": 0.04741950798779726, "clip_ratio/low_min": 0.04741950798779726, "clip_ratio/high_mean": 0.035813594702631235, "clip_ratio/high_max": 0.035813594702631235, "clip_ratio/region_mean": 0.0832331026904285, "reward_total_mean": 0.6040985584259033, "reward_meter_mean": 0.9604357481002808, "reward_meter_std": 0.0025081527419388294, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8992391228675842, "reward_repeat_soft_std": 0.05777277797460556, "reward_judge_quality_mean": 0.4312499761581421, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.6040985584259033, "reward_total_composite_std": 0.014271872118115425} {"timestamp_utc": "2026-04-13T11:59:26Z", "mode": "train", "global_step": 1887, "epoch": 0.18955298844801607, "loss": 0.0415, "grad_norm": 9.261543273925781, "learning_rate": 4.284848484848485e-06, "num_tokens": 3392507.0, "completions/mean_length": 44.625, "completions/min_length": 31.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.625, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.6726040244102478, "rewards/meter/std": 0.40645524859428406, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.790650486946106, "rewards/repeat_soft/std": 0.07165295630693436, "rewards/judge_quality/mean": 0.4099999964237213, "rewards/judge_quality/std": 0.06633248925209045, "rewards/total_composite/mean": 0.4995495676994324, "rewards/total_composite/std": 0.11973252892494202, "reward": 0.4995495676994324, "reward_std": 0.11973252147436142, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07547628879547119, "sampling/sampling_logp_difference/max": 1.427231788635254, "sampling/importance_sampling_ratio/min": 0.2399723082780838, "sampling/importance_sampling_ratio/mean": 1.0029417276382446, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.39059338346123695, "clip_ratio/low_mean": 0.027143463492393494, "clip_ratio/low_min": 0.027143463492393494, "clip_ratio/high_mean": 0.04254913330078125, "clip_ratio/high_max": 0.04254913330078125, "clip_ratio/region_mean": 0.06969259679317474, "reward_total_mean": 0.4995495676994324, "reward_meter_mean": 0.6726040244102478, "reward_meter_std": 0.40645524859428406, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.790650486946106, "reward_repeat_soft_std": 0.07165295630693436, "reward_judge_quality_mean": 0.4099999964237213, "reward_judge_quality_std": 0.06633248925209045, "reward_total_composite_mean": 0.4995495676994324, "reward_total_composite_std": 0.11973252892494202} {"timestamp_utc": "2026-04-13T11:59:34Z", "mode": "train", "global_step": 1888, "epoch": 0.18965344048216976, "loss": 0.0792, "grad_norm": 7.557681083679199, "learning_rate": 4.281818181818182e-06, "num_tokens": 3394379.0, "completions/mean_length": 72.0, "completions/min_length": 55.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.0, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.8011589646339417, "rewards/meter/std": 0.33095604181289673, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.830003023147583, "rewards/repeat_soft/std": 0.04706054553389549, "rewards/judge_quality/mean": 0.41499999165534973, "rewards/judge_quality/std": 0.14520922303199768, "rewards/total_composite/mean": 0.532202959060669, "rewards/total_composite/std": 0.1358167976140976, "reward": 0.532202959060669, "reward_std": 0.1358168125152588, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11180947721004486, "sampling/sampling_logp_difference/max": 1.8365421295166016, "sampling/importance_sampling_ratio/min": 0.15936754643917084, "sampling/importance_sampling_ratio/mean": 1.0054874420166016, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.697334386408329, "clip_ratio/low_mean": 0.044057718478143215, "clip_ratio/low_min": 0.044057718478143215, "clip_ratio/high_mean": 0.037179597187787294, "clip_ratio/high_max": 0.037179597187787294, "clip_ratio/region_mean": 0.08123731566593051, "reward_total_mean": 0.532202959060669, "reward_meter_mean": 0.8011589646339417, "reward_meter_std": 0.33095604181289673, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.830003023147583, "reward_repeat_soft_std": 0.04706054553389549, "reward_judge_quality_mean": 0.41499999165534973, "reward_judge_quality_std": 0.14520922303199768, "reward_total_composite_mean": 0.532202959060669, "reward_total_composite_std": 0.1358167976140976} {"timestamp_utc": "2026-04-13T11:59:45Z", "mode": "train", "global_step": 1889, "epoch": 0.18975389251632346, "loss": -0.1033, "grad_norm": 3.4962470531463623, "learning_rate": 4.278787878787879e-06, "num_tokens": 3395937.0, "completions/mean_length": 96.75, "completions/min_length": 29.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 37.42857360839844, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.5064377784729004, "rewards/meter/std": 0.45250165462493896, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9447498321533203, "rewards/repeat_soft/std": 0.08484742045402527, "rewards/judge_quality/mean": 0.3737499713897705, "rewards/judge_quality/std": 0.13081474602222443, "rewards/total_composite/mean": 0.436312198638916, "rewards/total_composite/std": 0.20593100786209106, "reward": 0.436312198638916, "reward_std": 0.20593099296092987, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15999017655849457, "sampling/sampling_logp_difference/max": 1.2710683345794678, "sampling/importance_sampling_ratio/min": 0.28053176403045654, "sampling/importance_sampling_ratio/mean": 1.0161256790161133, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.784225694835186, "clip_ratio/low_mean": 0.053742311894893646, "clip_ratio/low_min": 0.053742311894893646, "clip_ratio/high_mean": 0.07266347017139196, "clip_ratio/high_max": 0.07266347017139196, "clip_ratio/region_mean": 0.1264057820662856, "reward_total_mean": 0.436312198638916, "reward_meter_mean": 0.5064377784729004, "reward_meter_std": 0.45250165462493896, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9447498321533203, "reward_repeat_soft_std": 0.08484742045402527, "reward_judge_quality_mean": 0.3737499713897705, "reward_judge_quality_std": 0.13081474602222443, "reward_total_composite_mean": 0.436312198638916, "reward_total_composite_std": 0.20593100786209106} {"timestamp_utc": "2026-04-13T11:59:53Z", "mode": "train", "global_step": 1890, "epoch": 0.18985434455047714, "loss": -0.0124, "grad_norm": 5.567215442657471, "learning_rate": 4.275757575757576e-06, "num_tokens": 3398548.0, "completions/mean_length": 138.375, "completions/min_length": 122.0, "completions/max_length": 149.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 138.375, "completions/min_terminated_length": 122.0, "completions/max_terminated_length": 149.0, "rewards/meter/mean": 0.9934046268463135, "rewards/meter/std": 0.0010323392925783992, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7519139051437378, "rewards/repeat_soft/std": 0.0979333445429802, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.5743303298950195, "rewards/total_composite/std": 0.12184125930070877, "reward": 0.5743303298950195, "reward_std": 0.12184125185012817, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06597357243299484, "sampling/sampling_logp_difference/max": 1.602780818939209, "sampling/importance_sampling_ratio/min": 0.2013358473777771, "sampling/importance_sampling_ratio/mean": 1.0086263418197632, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.35575790889561176, "clip_ratio/low_mean": 0.04292728519067168, "clip_ratio/low_min": 0.04292728519067168, "clip_ratio/high_mean": 0.014261744916439056, "clip_ratio/high_max": 0.014261744916439056, "clip_ratio/region_mean": 0.05718903010711074, "reward_total_mean": 0.5743303298950195, "reward_meter_mean": 0.9934046268463135, "reward_meter_std": 0.0010323392925783992, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7519139051437378, "reward_repeat_soft_std": 0.0979333445429802, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.5743303298950195, "reward_total_composite_std": 0.12184125930070877} {"timestamp_utc": "2026-04-13T11:59:59Z", "mode": "train", "global_step": 1891, "epoch": 0.18995479658463083, "loss": 0.0069, "grad_norm": 11.446745872497559, "learning_rate": 4.272727272727273e-06, "num_tokens": 3400145.0, "completions/mean_length": 33.625, "completions/min_length": 29.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.625, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.9276425838470459, "rewards/meter/std": 0.09783067554235458, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8701931238174438, "rewards/repeat_soft/std": 0.053935930132865906, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.5884767770767212, "rewards/total_composite/std": 0.03028133139014244, "reward": 0.5884767770767212, "reward_std": 0.03028133697807789, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08166030049324036, "sampling/sampling_logp_difference/max": 1.5011775493621826, "sampling/importance_sampling_ratio/min": 0.290174663066864, "sampling/importance_sampling_ratio/mean": 0.9986169338226318, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.30929396115243435, "clip_ratio/low_mean": 0.007195723708719015, "clip_ratio/low_min": 0.007195723708719015, "clip_ratio/high_mean": 0.03407761291600764, "clip_ratio/high_max": 0.03407761291600764, "clip_ratio/region_mean": 0.04127333662472665, "reward_total_mean": 0.5884767770767212, "reward_meter_mean": 0.9276425838470459, "reward_meter_std": 0.09783067554235458, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8701931238174438, "reward_repeat_soft_std": 0.053935930132865906, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.5884767770767212, "reward_total_composite_std": 0.03028133139014244} {"timestamp_utc": "2026-04-13T12:00:05Z", "mode": "train", "global_step": 1892, "epoch": 0.19005524861878453, "loss": 0.1411, "grad_norm": 9.459722518920898, "learning_rate": 4.2696969696969695e-06, "num_tokens": 3402135.0, "completions/mean_length": 78.75, "completions/min_length": 49.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.75, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.8605332374572754, "rewards/meter/std": 0.20299570262432098, "rewards/count_adherence/mean": 0.7916666865348816, "rewards/count_adherence/std": 0.17251639068126678, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8585145473480225, "rewards/repeat_soft/std": 0.030557211488485336, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.5244558453559875, "rewards/total_composite/std": 0.06855473667383194, "reward": 0.5244558453559875, "reward_std": 0.06855473667383194, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09535474330186844, "sampling/sampling_logp_difference/max": 3.2019381523132324, "sampling/importance_sampling_ratio/min": 0.04068327695131302, "sampling/importance_sampling_ratio/mean": 1.0037459135055542, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.468066081404686, "clip_ratio/low_mean": 0.03889781702309847, "clip_ratio/low_min": 0.03889781702309847, "clip_ratio/high_mean": 0.039728786796331406, "clip_ratio/high_max": 0.039728786796331406, "clip_ratio/region_mean": 0.07862660381942987, "reward_total_mean": 0.5244558453559875, "reward_meter_mean": 0.8605332374572754, "reward_meter_std": 0.20299570262432098, "reward_count_adherence_mean": 0.7916666865348816, "reward_count_adherence_std": 0.17251639068126678, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8585145473480225, "reward_repeat_soft_std": 0.030557211488485336, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.5244558453559875, "reward_total_composite_std": 0.06855473667383194} {"timestamp_utc": "2026-04-13T12:00:12Z", "mode": "train", "global_step": 1893, "epoch": 0.19015570065293821, "loss": 0.0269, "grad_norm": 10.69278621673584, "learning_rate": 4.266666666666668e-06, "num_tokens": 3403761.0, "completions/mean_length": 45.25, "completions/min_length": 41.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.25, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.9858983755111694, "rewards/meter/std": 0.004938791040331125, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9576983451843262, "rewards/repeat_soft/std": 0.021132754161953926, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.6529911756515503, "rewards/total_composite/std": 0.11349096894264221, "reward": 0.6529911756515503, "reward_std": 0.11349096894264221, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08822016417980194, "sampling/sampling_logp_difference/max": 1.7165651321411133, "sampling/importance_sampling_ratio/min": 0.17968228459358215, "sampling/importance_sampling_ratio/mean": 1.0064579248428345, "sampling/importance_sampling_ratio/max": 1.6938608884811401, "entropy": 0.4698624052107334, "clip_ratio/low_mean": 0.06939548999071121, "clip_ratio/low_min": 0.06939548999071121, "clip_ratio/high_mean": 0.011363636702299118, "clip_ratio/high_max": 0.011363636702299118, "clip_ratio/region_mean": 0.08075912669301033, "reward_total_mean": 0.6529911756515503, "reward_meter_mean": 0.9858983755111694, "reward_meter_std": 0.004938791040331125, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9576983451843262, "reward_repeat_soft_std": 0.021132754161953926, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.6529911756515503, "reward_total_composite_std": 0.11349096894264221} {"timestamp_utc": "2026-04-13T12:00:19Z", "mode": "train", "global_step": 1894, "epoch": 0.19025615268709192, "loss": 0.0732, "grad_norm": 5.861888885498047, "learning_rate": 4.263636363636364e-06, "num_tokens": 3406006.0, "completions/mean_length": 112.625, "completions/min_length": 79.0, "completions/max_length": 144.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.625, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 144.0, "rewards/meter/mean": 0.9630520939826965, "rewards/meter/std": 0.07279664278030396, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.13363061845302582, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7933350801467896, "rewards/repeat_soft/std": 0.11016028374433517, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.5295034646987915, "rewards/total_composite/std": 0.055732253938913345, "reward": 0.5295034646987915, "reward_std": 0.05573224276304245, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09980825334787369, "sampling/sampling_logp_difference/max": 4.133489608764648, "sampling/importance_sampling_ratio/min": 0.016026854515075684, "sampling/importance_sampling_ratio/mean": 1.0049759149551392, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4927608408033848, "clip_ratio/low_mean": 0.02559055434539914, "clip_ratio/low_min": 0.02559055434539914, "clip_ratio/high_mean": 0.07266552560031414, "clip_ratio/high_max": 0.07266552560031414, "clip_ratio/region_mean": 0.09825607994571328, "reward_total_mean": 0.5295034646987915, "reward_meter_mean": 0.9630520939826965, "reward_meter_std": 0.07279664278030396, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.13363061845302582, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7933350801467896, "reward_repeat_soft_std": 0.11016028374433517, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.5295034646987915, "reward_total_composite_std": 0.055732253938913345} {"timestamp_utc": "2026-04-13T12:00:27Z", "mode": "train", "global_step": 1895, "epoch": 0.1903566047212456, "loss": 0.0533, "grad_norm": 11.755945205688477, "learning_rate": 4.260606060606061e-06, "num_tokens": 3408047.0, "completions/mean_length": 89.125, "completions/min_length": 77.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.125, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.5747484564781189, "rewards/meter/std": 0.36859145760536194, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.13363061845302582, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9119118452072144, "rewards/repeat_soft/std": 0.03609064221382141, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.4490486979484558, "rewards/total_composite/std": 0.10694235563278198, "reward": 0.4490486979484558, "reward_std": 0.10694236308336258, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14104905724525452, "sampling/sampling_logp_difference/max": 3.7344534397125244, "sampling/importance_sampling_ratio/min": 0.023886224254965782, "sampling/importance_sampling_ratio/mean": 0.983735203742981, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.559051401913166, "clip_ratio/low_mean": 0.04991707857698202, "clip_ratio/low_min": 0.04991707857698202, "clip_ratio/high_mean": 0.06310982350260019, "clip_ratio/high_max": 0.06310982350260019, "clip_ratio/region_mean": 0.11302690207958221, "reward_total_mean": 0.4490486979484558, "reward_meter_mean": 0.5747484564781189, "reward_meter_std": 0.36859145760536194, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.13363061845302582, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9119118452072144, "reward_repeat_soft_std": 0.03609064221382141, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.4490486979484558, "reward_total_composite_std": 0.10694235563278198} {"timestamp_utc": "2026-04-13T12:00:38Z", "mode": "train", "global_step": 1896, "epoch": 0.19045705675539928, "loss": -0.0977, "grad_norm": 1.9929245710372925, "learning_rate": 4.2575757575757585e-06, "num_tokens": 3409574.0, "completions/mean_length": 92.875, "completions/min_length": 26.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 33.0, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.8006120324134827, "rewards/meter/std": 0.2018144726753235, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9413082003593445, "rewards/repeat_soft/std": 0.053778115659952164, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.13274572789669037, "rewards/total_composite/mean": 0.4947419762611389, "rewards/total_composite/std": 0.20789597928524017, "reward": 0.4947419762611389, "reward_std": 0.20789596438407898, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14270998537540436, "sampling/sampling_logp_difference/max": 1.1320117712020874, "sampling/importance_sampling_ratio/min": 0.3223840296268463, "sampling/importance_sampling_ratio/mean": 1.0156162977218628, "sampling/importance_sampling_ratio/max": 1.8814842700958252, "entropy": 0.9908416122198105, "clip_ratio/low_mean": 0.007575757801532745, "clip_ratio/low_min": 0.007575757801532745, "clip_ratio/high_mean": 0.10267268819734454, "clip_ratio/high_max": 0.10267268819734454, "clip_ratio/region_mean": 0.11024844599887729, "reward_total_mean": 0.4947419762611389, "reward_meter_mean": 0.8006120324134827, "reward_meter_std": 0.2018144726753235, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9413082003593445, "reward_repeat_soft_std": 0.053778115659952164, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.13274572789669037, "reward_total_composite_mean": 0.4947419762611389, "reward_total_composite_std": 0.20789597928524017} {"timestamp_utc": "2026-04-13T12:00:44Z", "mode": "train", "global_step": 1897, "epoch": 0.190557508789553, "loss": -0.0055, "grad_norm": 21.212984085083008, "learning_rate": 4.254545454545455e-06, "num_tokens": 3410863.0, "completions/mean_length": 21.125, "completions/min_length": 17.0, "completions/max_length": 25.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.125, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 25.0, "rewards/meter/mean": 0.9038698673248291, "rewards/meter/std": 0.1721266359090805, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9348776340484619, "rewards/repeat_soft/std": 0.054454438388347626, "rewards/judge_quality/mean": 0.48250001668930054, "rewards/judge_quality/std": 0.1767767071723938, "rewards/total_composite/mean": 0.6266239881515503, "rewards/total_composite/std": 0.13112255930900574, "reward": 0.6266239881515503, "reward_std": 0.13112254440784454, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15069088339805603, "sampling/sampling_logp_difference/max": 1.5056264400482178, "sampling/importance_sampling_ratio/min": 0.34185534715652466, "sampling/importance_sampling_ratio/mean": 1.0511457920074463, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.155650481581688, "clip_ratio/low_mean": 0.12694388814270496, "clip_ratio/low_min": 0.12694388814270496, "clip_ratio/high_mean": 0.010869565419852734, "clip_ratio/high_max": 0.010869565419852734, "clip_ratio/region_mean": 0.1378134535625577, "reward_total_mean": 0.6266239881515503, "reward_meter_mean": 0.9038698673248291, "reward_meter_std": 0.1721266359090805, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9348776340484619, "reward_repeat_soft_std": 0.054454438388347626, "reward_judge_quality_mean": 0.48250001668930054, "reward_judge_quality_std": 0.1767767071723938, "reward_total_composite_mean": 0.6266239881515503, "reward_total_composite_std": 0.13112255930900574} {"timestamp_utc": "2026-04-13T12:00:52Z", "mode": "train", "global_step": 1898, "epoch": 0.19065796082370667, "loss": 0.0936, "grad_norm": 6.053118705749512, "learning_rate": 4.251515151515152e-06, "num_tokens": 3413535.0, "completions/mean_length": 146.0, "completions/min_length": 126.0, "completions/max_length": 173.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 146.0, "completions/min_terminated_length": 126.0, "completions/max_terminated_length": 173.0, "rewards/meter/mean": 0.7404102087020874, "rewards/meter/std": 0.2980844974517822, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.07715168595314026, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8465932607650757, "rewards/repeat_soft/std": 0.0765487477183342, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.528233528137207, "rewards/total_composite/std": 0.10594785958528519, "reward": 0.528233528137207, "reward_std": 0.10594785958528519, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09265933930873871, "sampling/sampling_logp_difference/max": 1.8282966613769531, "sampling/importance_sampling_ratio/min": 0.16068704426288605, "sampling/importance_sampling_ratio/mean": 1.0120998620986938, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.529948141425848, "clip_ratio/low_mean": 0.03747152350842953, "clip_ratio/low_min": 0.03747152350842953, "clip_ratio/high_mean": 0.030135025968775153, "clip_ratio/high_max": 0.030135025968775153, "clip_ratio/region_mean": 0.06760654947720468, "reward_total_mean": 0.528233528137207, "reward_meter_mean": 0.7404102087020874, "reward_meter_std": 0.2980844974517822, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.07715168595314026, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8465932607650757, "reward_repeat_soft_std": 0.0765487477183342, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.528233528137207, "reward_total_composite_std": 0.10594785958528519} {"timestamp_utc": "2026-04-13T12:00:58Z", "mode": "train", "global_step": 1899, "epoch": 0.19075841285786038, "loss": 0.1288, "grad_norm": 13.905967712402344, "learning_rate": 4.248484848484849e-06, "num_tokens": 3415019.0, "completions/mean_length": 41.5, "completions/min_length": 33.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.5, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.8378007411956787, "rewards/meter/std": 0.32178792357444763, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9815511107444763, "rewards/repeat_soft/std": 0.011156021617352962, "rewards/judge_quality/mean": 0.42374998331069946, "rewards/judge_quality/std": 0.1524970829486847, "rewards/total_composite/mean": 0.5784533023834229, "rewards/total_composite/std": 0.13378991186618805, "reward": 0.5784533023834229, "reward_std": 0.13378991186618805, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13036690652370453, "sampling/sampling_logp_difference/max": 1.4765377044677734, "sampling/importance_sampling_ratio/min": 0.2284272164106369, "sampling/importance_sampling_ratio/mean": 1.0019536018371582, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8005569577217102, "clip_ratio/low_mean": 0.03787206578999758, "clip_ratio/low_min": 0.03787206578999758, "clip_ratio/high_mean": 0.1173378536477685, "clip_ratio/high_max": 0.1173378536477685, "clip_ratio/region_mean": 0.15520991943776608, "reward_total_mean": 0.5784533023834229, "reward_meter_mean": 0.8378007411956787, "reward_meter_std": 0.32178792357444763, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9815511107444763, "reward_repeat_soft_std": 0.011156021617352962, "reward_judge_quality_mean": 0.42374998331069946, "reward_judge_quality_std": 0.1524970829486847, "reward_total_composite_mean": 0.5784533023834229, "reward_total_composite_std": 0.13378991186618805} {"timestamp_utc": "2026-04-13T12:01:04Z", "mode": "train", "global_step": 1900, "epoch": 0.19085886489201406, "loss": -0.0636, "grad_norm": 12.112760543823242, "learning_rate": 4.245454545454546e-06, "num_tokens": 3416585.0, "completions/mean_length": 41.75, "completions/min_length": 32.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.75, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.8482105135917664, "rewards/meter/std": 0.23347781598567963, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9181222915649414, "rewards/repeat_soft/std": 0.0497940294444561, "rewards/judge_quality/mean": 0.7950000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.7733848094940186, "rewards/total_composite/std": 0.18588495254516602, "reward": 0.7733848094940186, "reward_std": 0.18588493764400482, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14247700572013855, "sampling/sampling_logp_difference/max": 0.990262508392334, "sampling/importance_sampling_ratio/min": 0.38564333319664, "sampling/importance_sampling_ratio/mean": 1.0247104167938232, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1424989625811577, "clip_ratio/low_mean": 0.056029041297733784, "clip_ratio/low_min": 0.056029041297733784, "clip_ratio/high_mean": 0.08261819859035313, "clip_ratio/high_max": 0.08261819859035313, "clip_ratio/region_mean": 0.13864723988808692, "reward_total_mean": 0.7733848094940186, "reward_meter_mean": 0.8482105135917664, "reward_meter_std": 0.23347781598567963, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9181222915649414, "reward_repeat_soft_std": 0.0497940294444561, "reward_judge_quality_mean": 0.7950000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.7733848094940186, "reward_total_composite_std": 0.18588495254516602} {"timestamp_utc": "2026-04-13T12:01:47Z", "mode": "eval", "global_step": 1900, "epoch": 0.19085886489201406, "eval_loss": NaN, "eval_runtime": 42.3837, "eval_samples_per_second": 1.888, "eval_steps_per_second": 0.236, "eval_num_tokens": 3416585.0, "eval_completions/mean_length": 78.225, "eval_completions/min_length": 32.6, "eval_completions/max_length": 141.3, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 78.225, "eval_completions/min_terminated_length": 32.6, "eval_completions/max_terminated_length": 141.3, "eval_rewards/meter/mean": 0.8480098903179168, "eval_rewards/meter/std": 0.23681166246533394, "eval_rewards/count_adherence/mean": 0.9772916555404663, "eval_rewards/count_adherence/std": 0.05044138841331005, "eval_rewards/hard_gate/mean": 1.0, "eval_rewards/hard_gate/std": 0.0, "eval_rewards/repeat_soft/mean": 0.8693194091320038, "eval_rewards/repeat_soft/std": 0.08963314779102802, "eval_rewards/judge_quality/mean": 0.46762500107288363, "eval_rewards/judge_quality/std": 0.12215339820832014, "eval_rewards/total_composite/mean": 0.5825744986534118, "eval_rewards/total_composite/std": 0.10912723653018475, "eval_reward": 0.5825744986534118, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.05554373003542423, "eval_sampling/sampling_logp_difference/max": 0.9246527910232544, "eval_sampling/importance_sampling_ratio/min": 0.40136079490184784, "eval_sampling/importance_sampling_ratio/mean": 1.0137291550636292, "eval_sampling/importance_sampling_ratio/max": 1.4351726770401, "eval_entropy": 0.5898070454597473, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.5825744986534118, "eval_reward_meter_mean": 0.8480098903179168, "eval_reward_meter_std": 0.23681166246533394, "eval_reward_count_adherence_mean": 0.9772916555404663, "eval_reward_count_adherence_std": 0.05044138841331005, "eval_reward_hard_gate_mean": 1.0, "eval_reward_hard_gate_std": 0.0, "eval_reward_repeat_soft_mean": 0.8693194091320038, "eval_reward_repeat_soft_std": 0.08963314779102802, "eval_reward_judge_quality_mean": 0.46762500107288363, "eval_reward_judge_quality_std": 0.12215339820832014, "eval_reward_total_composite_mean": 0.5825744986534118, "eval_reward_total_composite_std": 0.10912723653018475} {"timestamp_utc": "2026-04-13T12:02:02Z", "mode": "train", "global_step": 1901, "epoch": 0.19095931692616774, "loss": -0.1149, "grad_norm": 1.9661214351654053, "learning_rate": 4.242424242424243e-06, "num_tokens": 3418105.0, "completions/mean_length": 162.0, "completions/min_length": 35.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 45.333335876464844, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.6698601245880127, "rewards/meter/std": 0.41053760051727295, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9652444124221802, "rewards/repeat_soft/std": 0.03090461529791355, "rewards/judge_quality/mean": 0.45249998569488525, "rewards/judge_quality/std": 0.33065950870513916, "rewards/total_composite/mean": 0.5056458115577698, "rewards/total_composite/std": 0.36281538009643555, "reward": 0.5056458115577698, "reward_std": 0.36281538009643555, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11610380560159683, "sampling/sampling_logp_difference/max": 1.0769240856170654, "sampling/importance_sampling_ratio/min": 0.34064167737960815, "sampling/importance_sampling_ratio/mean": 1.004833459854126, "sampling/importance_sampling_ratio/max": 1.8720961809158325, "entropy": 0.564405545592308, "clip_ratio/low_mean": 0.01785714365541935, "clip_ratio/low_min": 0.01785714365541935, "clip_ratio/high_mean": 0.0837968117557466, "clip_ratio/high_max": 0.0837968117557466, "clip_ratio/region_mean": 0.10165395541116595, "reward_total_mean": 0.5056458115577698, "reward_meter_mean": 0.6698601245880127, "reward_meter_std": 0.41053760051727295, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9652444124221802, "reward_repeat_soft_std": 0.03090461529791355, "reward_judge_quality_mean": 0.45249998569488525, "reward_judge_quality_std": 0.33065950870513916, "reward_total_composite_mean": 0.5056458115577698, "reward_total_composite_std": 0.36281538009643555} {"timestamp_utc": "2026-04-13T12:02:14Z", "mode": "train", "global_step": 1902, "epoch": 0.19105976896032145, "loss": -0.066, "grad_norm": 4.033031463623047, "learning_rate": 4.2393939393939395e-06, "num_tokens": 3419486.0, "completions/mean_length": 84.625, "completions/min_length": 16.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 23.571430206298828, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.7202394008636475, "rewards/meter/std": 0.4406243860721588, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.945844292640686, "rewards/repeat_soft/std": 0.04052875190973282, "rewards/judge_quality/mean": 0.7524999976158142, "rewards/judge_quality/std": 0.3345252573490143, "rewards/total_composite/mean": 0.6909451484680176, "rewards/total_composite/std": 0.3505527675151825, "reward": 0.6909451484680176, "reward_std": 0.3505527675151825, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18114034831523895, "sampling/sampling_logp_difference/max": 1.3069181442260742, "sampling/importance_sampling_ratio/min": 0.30163708329200745, "sampling/importance_sampling_ratio/mean": 1.0173735618591309, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0979121401906013, "clip_ratio/low_mean": 0.0633680559694767, "clip_ratio/low_min": 0.0633680559694767, "clip_ratio/high_mean": 0.14102468639612198, "clip_ratio/high_max": 0.14102468639612198, "clip_ratio/region_mean": 0.20439274236559868, "reward_total_mean": 0.6909451484680176, "reward_meter_mean": 0.7202394008636475, "reward_meter_std": 0.4406243860721588, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.945844292640686, "reward_repeat_soft_std": 0.04052875190973282, "reward_judge_quality_mean": 0.7524999976158142, "reward_judge_quality_std": 0.3345252573490143, "reward_total_composite_mean": 0.6909451484680176, "reward_total_composite_std": 0.3505527675151825} {"timestamp_utc": "2026-04-13T12:02:20Z", "mode": "train", "global_step": 1903, "epoch": 0.19116022099447513, "loss": -0.0389, "grad_norm": 14.084986686706543, "learning_rate": 4.236363636363637e-06, "num_tokens": 3421074.0, "completions/mean_length": 25.5, "completions/min_length": 18.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.5, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.6613826751708984, "rewards/meter/std": 0.43064865469932556, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8781934976577759, "rewards/repeat_soft/std": 0.09914916008710861, "rewards/judge_quality/mean": 0.49000000953674316, "rewards/judge_quality/std": 0.1742740124464035, "rewards/total_composite/mean": 0.5169066190719604, "rewards/total_composite/std": 0.1330171823501587, "reward": 0.5169066190719604, "reward_std": 0.13301719725131989, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11851590126752853, "sampling/sampling_logp_difference/max": 1.0182857513427734, "sampling/importance_sampling_ratio/min": 0.3612136244773865, "sampling/importance_sampling_ratio/mean": 1.0146753787994385, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7701932080090046, "clip_ratio/low_mean": 0.034722222946584225, "clip_ratio/low_min": 0.034722222946584225, "clip_ratio/high_mean": 0.08670286927372217, "clip_ratio/high_max": 0.08670286927372217, "clip_ratio/region_mean": 0.1214250922203064, "reward_total_mean": 0.5169066190719604, "reward_meter_mean": 0.6613826751708984, "reward_meter_std": 0.43064865469932556, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8781934976577759, "reward_repeat_soft_std": 0.09914916008710861, "reward_judge_quality_mean": 0.49000000953674316, "reward_judge_quality_std": 0.1742740124464035, "reward_total_composite_mean": 0.5169066190719604, "reward_total_composite_std": 0.1330171823501587} {"timestamp_utc": "2026-04-13T12:02:27Z", "mode": "train", "global_step": 1904, "epoch": 0.19126067302862884, "loss": 0.0592, "grad_norm": 10.84548282623291, "learning_rate": 4.233333333333334e-06, "num_tokens": 3422843.0, "completions/mean_length": 51.125, "completions/min_length": 43.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.125, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.6600973606109619, "rewards/meter/std": 0.31269407272338867, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9792370796203613, "rewards/repeat_soft/std": 0.014737443998456001, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.5339018106460571, "rewards/total_composite/std": 0.09232258051633835, "reward": 0.5339018106460571, "reward_std": 0.09232258796691895, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15003661811351776, "sampling/sampling_logp_difference/max": 2.6863467693328857, "sampling/importance_sampling_ratio/min": 0.06812937557697296, "sampling/importance_sampling_ratio/mean": 1.0140804052352905, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9993136450648308, "clip_ratio/low_mean": 0.054302156902849674, "clip_ratio/low_min": 0.054302156902849674, "clip_ratio/high_mean": 0.10764239821583033, "clip_ratio/high_max": 0.10764239821583033, "clip_ratio/region_mean": 0.16194455511868, "reward_total_mean": 0.5339018106460571, "reward_meter_mean": 0.6600973606109619, "reward_meter_std": 0.31269407272338867, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9792370796203613, "reward_repeat_soft_std": 0.014737443998456001, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.5339018106460571, "reward_total_composite_std": 0.09232258051633835} {"timestamp_utc": "2026-04-13T12:02:38Z", "mode": "train", "global_step": 1905, "epoch": 0.19136112506278252, "loss": -0.1045, "grad_norm": 2.3378593921661377, "learning_rate": 4.2303030303030304e-06, "num_tokens": 3424337.0, "completions/mean_length": 96.75, "completions/min_length": 32.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 37.42857360839844, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.8729835748672485, "rewards/meter/std": 0.30031493306159973, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8887836933135986, "rewards/repeat_soft/std": 0.07469533383846283, "rewards/judge_quality/mean": 0.4399999976158142, "rewards/judge_quality/std": 0.23439893126487732, "rewards/total_composite/mean": 0.5661696791648865, "rewards/total_composite/std": 0.2565822899341583, "reward": 0.5661696791648865, "reward_std": 0.2565822899341583, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10371766239404678, "sampling/sampling_logp_difference/max": 1.1465868949890137, "sampling/importance_sampling_ratio/min": 0.31771931052207947, "sampling/importance_sampling_ratio/mean": 1.0026928186416626, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5314153209328651, "clip_ratio/low_mean": 0.015625, "clip_ratio/low_min": 0.015625, "clip_ratio/high_mean": 0.07597637944854796, "clip_ratio/high_max": 0.07597637944854796, "clip_ratio/region_mean": 0.09160137944854796, "reward_total_mean": 0.5661696791648865, "reward_meter_mean": 0.8729835748672485, "reward_meter_std": 0.30031493306159973, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8887836933135986, "reward_repeat_soft_std": 0.07469533383846283, "reward_judge_quality_mean": 0.4399999976158142, "reward_judge_quality_std": 0.23439893126487732, "reward_total_composite_mean": 0.5661696791648865, "reward_total_composite_std": 0.2565822899341583} {"timestamp_utc": "2026-04-13T12:02:45Z", "mode": "train", "global_step": 1906, "epoch": 0.1914615770969362, "loss": 0.04, "grad_norm": 12.90833854675293, "learning_rate": 4.227272727272728e-06, "num_tokens": 3425960.0, "completions/mean_length": 53.875, "completions/min_length": 45.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.875, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.8947017192840576, "rewards/meter/std": 0.2500592768192291, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9868673086166382, "rewards/repeat_soft/std": 0.01419753022491932, "rewards/judge_quality/mean": 0.4937500059604645, "rewards/judge_quality/std": 0.1728696972131729, "rewards/total_composite/mean": 0.6106759309768677, "rewards/total_composite/std": 0.03996795788407326, "reward": 0.6106759309768677, "reward_std": 0.039967965334653854, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12730588018894196, "sampling/sampling_logp_difference/max": 2.330036163330078, "sampling/importance_sampling_ratio/min": 0.09729223698377609, "sampling/importance_sampling_ratio/mean": 1.0107167959213257, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7010943591594696, "clip_ratio/low_mean": 0.008620689623057842, "clip_ratio/low_min": 0.008620689623057842, "clip_ratio/high_mean": 0.09743543481454253, "clip_ratio/high_max": 0.09743543481454253, "clip_ratio/region_mean": 0.10605612443760037, "reward_total_mean": 0.6106759309768677, "reward_meter_mean": 0.8947017192840576, "reward_meter_std": 0.2500592768192291, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9868673086166382, "reward_repeat_soft_std": 0.01419753022491932, "reward_judge_quality_mean": 0.4937500059604645, "reward_judge_quality_std": 0.1728696972131729, "reward_total_composite_mean": 0.6106759309768677, "reward_total_composite_std": 0.03996795788407326} {"timestamp_utc": "2026-04-13T12:02:56Z", "mode": "train", "global_step": 1907, "epoch": 0.1915620291310899, "loss": -0.0754, "grad_norm": 1.5282032489776611, "learning_rate": 4.224242424242425e-06, "num_tokens": 3427283.0, "completions/mean_length": 148.375, "completions/min_length": 20.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 27.166667938232422, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.6922097206115723, "rewards/meter/std": 0.35771870613098145, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9556818008422852, "rewards/repeat_soft/std": 0.01928471215069294, "rewards/judge_quality/mean": 0.3137499690055847, "rewards/judge_quality/std": 0.17492344975471497, "rewards/total_composite/mean": 0.3942892253398895, "rewards/total_composite/std": 0.25641167163848877, "reward": 0.3942892253398895, "reward_std": 0.25641167163848877, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17665165662765503, "sampling/sampling_logp_difference/max": 1.8716812133789062, "sampling/importance_sampling_ratio/min": 0.1538647562265396, "sampling/importance_sampling_ratio/mean": 1.0526965856552124, "sampling/importance_sampling_ratio/max": 1.782524824142456, "entropy": 1.2915838807821274, "clip_ratio/low_mean": 0.011904762126505375, "clip_ratio/low_min": 0.011904762126505375, "clip_ratio/high_mean": 0.10561077040620148, "clip_ratio/high_max": 0.10561077040620148, "clip_ratio/region_mean": 0.11751553253270686, "reward_total_mean": 0.3942892253398895, "reward_meter_mean": 0.6922097206115723, "reward_meter_std": 0.35771870613098145, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9556818008422852, "reward_repeat_soft_std": 0.01928471215069294, "reward_judge_quality_mean": 0.3137499690055847, "reward_judge_quality_std": 0.17492344975471497, "reward_total_composite_mean": 0.3942892253398895, "reward_total_composite_std": 0.25641167163848877} {"timestamp_utc": "2026-04-13T12:03:03Z", "mode": "train", "global_step": 1908, "epoch": 0.1916624811652436, "loss": 0.0469, "grad_norm": 11.930583953857422, "learning_rate": 4.221212121212121e-06, "num_tokens": 3429356.0, "completions/mean_length": 85.125, "completions/min_length": 76.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.125, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.5915040969848633, "rewards/meter/std": 0.2694706320762634, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.982235312461853, "rewards/repeat_soft/std": 0.009068382903933525, "rewards/judge_quality/mean": 0.5199999809265137, "rewards/judge_quality/std": 0.19272483885288239, "rewards/total_composite/mean": 0.5378624200820923, "rewards/total_composite/std": 0.08761601895093918, "reward": 0.5378624200820923, "reward_std": 0.08761601150035858, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1465836614370346, "sampling/sampling_logp_difference/max": 1.7468070983886719, "sampling/importance_sampling_ratio/min": 0.17432966828346252, "sampling/importance_sampling_ratio/mean": 0.9981339573860168, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7508384957909584, "clip_ratio/low_mean": 0.06788656301796436, "clip_ratio/low_min": 0.06788656301796436, "clip_ratio/high_mean": 0.06018355302512646, "clip_ratio/high_max": 0.06018355302512646, "clip_ratio/region_mean": 0.12807011604309082, "reward_total_mean": 0.5378624200820923, "reward_meter_mean": 0.5915040969848633, "reward_meter_std": 0.2694706320762634, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.982235312461853, "reward_repeat_soft_std": 0.009068382903933525, "reward_judge_quality_mean": 0.5199999809265137, "reward_judge_quality_std": 0.19272483885288239, "reward_total_composite_mean": 0.5378624200820923, "reward_total_composite_std": 0.08761601895093918} {"timestamp_utc": "2026-04-13T12:03:09Z", "mode": "train", "global_step": 1909, "epoch": 0.1917629331993973, "loss": 0.1083, "grad_norm": 11.530652046203613, "learning_rate": 4.218181818181819e-06, "num_tokens": 3431028.0, "completions/mean_length": 46.0, "completions/min_length": 41.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.0, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9786093831062317, "rewards/meter/std": 0.01941728964447975, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9682313203811646, "rewards/repeat_soft/std": 0.038088828325271606, "rewards/judge_quality/mean": 0.5850000381469727, "rewards/judge_quality/std": 0.2946668863296509, "rewards/total_composite/mean": 0.7170112133026123, "rewards/total_composite/std": 0.18612056970596313, "reward": 0.7170112133026123, "reward_std": 0.18612055480480194, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1269945502281189, "sampling/sampling_logp_difference/max": 1.3092458248138428, "sampling/importance_sampling_ratio/min": 0.3732983469963074, "sampling/importance_sampling_ratio/mean": 1.0166714191436768, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9716865196824074, "clip_ratio/low_mean": 0.0770115265622735, "clip_ratio/low_min": 0.0770115265622735, "clip_ratio/high_mean": 0.05192000512033701, "clip_ratio/high_max": 0.05192000512033701, "clip_ratio/region_mean": 0.1289315316826105, "reward_total_mean": 0.7170112133026123, "reward_meter_mean": 0.9786093831062317, "reward_meter_std": 0.01941728964447975, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9682313203811646, "reward_repeat_soft_std": 0.038088828325271606, "reward_judge_quality_mean": 0.5850000381469727, "reward_judge_quality_std": 0.2946668863296509, "reward_total_composite_mean": 0.7170112133026123, "reward_total_composite_std": 0.18612056970596313} {"timestamp_utc": "2026-04-13T12:03:16Z", "mode": "train", "global_step": 1910, "epoch": 0.19186338523355098, "loss": 0.0465, "grad_norm": 6.4757304191589355, "learning_rate": 4.215151515151515e-06, "num_tokens": 3433133.0, "completions/mean_length": 103.125, "completions/min_length": 76.0, "completions/max_length": 133.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 103.125, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.9347220063209534, "rewards/meter/std": 0.1163896769285202, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8798757791519165, "rewards/repeat_soft/std": 0.07860218733549118, "rewards/judge_quality/mean": 0.45749998092651367, "rewards/judge_quality/std": 0.10606604069471359, "rewards/total_composite/mean": 0.6112393736839294, "rewards/total_composite/std": 0.08730139583349228, "reward": 0.6112393736839294, "reward_std": 0.08730139583349228, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11001424491405487, "sampling/sampling_logp_difference/max": 1.6104984283447266, "sampling/importance_sampling_ratio/min": 0.19978800415992737, "sampling/importance_sampling_ratio/mean": 1.0204784870147705, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8050570487976074, "clip_ratio/low_mean": 0.06305000837892294, "clip_ratio/low_min": 0.06305000837892294, "clip_ratio/high_mean": 0.03493107855319977, "clip_ratio/high_max": 0.03493107855319977, "clip_ratio/region_mean": 0.09798108693212271, "reward_total_mean": 0.6112393736839294, "reward_meter_mean": 0.9347220063209534, "reward_meter_std": 0.1163896769285202, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8798757791519165, "reward_repeat_soft_std": 0.07860218733549118, "reward_judge_quality_mean": 0.45749998092651367, "reward_judge_quality_std": 0.10606604069471359, "reward_total_composite_mean": 0.6112393736839294, "reward_total_composite_std": 0.08730139583349228} {"timestamp_utc": "2026-04-13T12:03:23Z", "mode": "train", "global_step": 1911, "epoch": 0.19196383726770466, "loss": 0.0613, "grad_norm": 6.657021522521973, "learning_rate": 4.212121212121212e-06, "num_tokens": 3435058.0, "completions/mean_length": 64.625, "completions/min_length": 55.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.625, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.954378068447113, "rewards/meter/std": 0.08675800263881683, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8774073719978333, "rewards/repeat_soft/std": 0.0204410869628191, "rewards/judge_quality/mean": 0.5362499952316284, "rewards/judge_quality/std": 0.1524970978498459, "rewards/total_composite/mean": 0.6668540239334106, "rewards/total_composite/std": 0.10783916711807251, "reward": 0.6668540239334106, "reward_std": 0.10783915966749191, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07967215776443481, "sampling/sampling_logp_difference/max": 1.5013203620910645, "sampling/importance_sampling_ratio/min": 0.2228357493877411, "sampling/importance_sampling_ratio/mean": 1.0027366876602173, "sampling/importance_sampling_ratio/max": 1.8892652988433838, "entropy": 0.4966052621603012, "clip_ratio/low_mean": 0.04203710425645113, "clip_ratio/low_min": 0.04203710425645113, "clip_ratio/high_mean": 0.03393878135830164, "clip_ratio/high_max": 0.03393878135830164, "clip_ratio/region_mean": 0.07597588561475277, "reward_total_mean": 0.6668540239334106, "reward_meter_mean": 0.954378068447113, "reward_meter_std": 0.08675800263881683, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8774073719978333, "reward_repeat_soft_std": 0.0204410869628191, "reward_judge_quality_mean": 0.5362499952316284, "reward_judge_quality_std": 0.1524970978498459, "reward_total_composite_mean": 0.6668540239334106, "reward_total_composite_std": 0.10783916711807251} {"timestamp_utc": "2026-04-13T12:03:31Z", "mode": "train", "global_step": 1912, "epoch": 0.19206428930185837, "loss": 0.0134, "grad_norm": 7.404308319091797, "learning_rate": 4.2090909090909095e-06, "num_tokens": 3437690.0, "completions/mean_length": 142.0, "completions/min_length": 129.0, "completions/max_length": 154.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 142.0, "completions/min_terminated_length": 129.0, "completions/max_terminated_length": 154.0, "rewards/meter/mean": 0.991479754447937, "rewards/meter/std": 0.003022808115929365, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7146003842353821, "rewards/repeat_soft/std": 0.1012619212269783, "rewards/judge_quality/mean": 0.2462500035762787, "rewards/judge_quality/std": 0.08348438143730164, "rewards/total_composite/mean": 0.4659019708633423, "rewards/total_composite/std": 0.060872793197631836, "reward": 0.4659019708633423, "reward_std": 0.06087278574705124, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05452094227075577, "sampling/sampling_logp_difference/max": 1.2917299270629883, "sampling/importance_sampling_ratio/min": 0.2747949957847595, "sampling/importance_sampling_ratio/mean": 1.0096286535263062, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3699205368757248, "clip_ratio/low_mean": 0.026885909494012594, "clip_ratio/low_min": 0.026885909494012594, "clip_ratio/high_mean": 0.023657962679862976, "clip_ratio/high_max": 0.023657962679862976, "clip_ratio/region_mean": 0.05054387217387557, "reward_total_mean": 0.4659019708633423, "reward_meter_mean": 0.991479754447937, "reward_meter_std": 0.003022808115929365, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7146003842353821, "reward_repeat_soft_std": 0.1012619212269783, "reward_judge_quality_mean": 0.2462500035762787, "reward_judge_quality_std": 0.08348438143730164, "reward_total_composite_mean": 0.4659019708633423, "reward_total_composite_std": 0.060872793197631836} {"timestamp_utc": "2026-04-13T12:03:38Z", "mode": "train", "global_step": 1913, "epoch": 0.19216474133601205, "loss": 0.0689, "grad_norm": 6.058612823486328, "learning_rate": 4.206060606060606e-06, "num_tokens": 3439994.0, "completions/mean_length": 108.0, "completions/min_length": 89.0, "completions/max_length": 124.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.0, "completions/min_terminated_length": 89.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.9075185656547546, "rewards/meter/std": 0.2047087401151657, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8628815412521362, "rewards/repeat_soft/std": 0.03367576375603676, "rewards/judge_quality/mean": 0.41999998688697815, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.577184796333313, "rewards/total_composite/std": 0.057645779103040695, "reward": 0.577184796333313, "reward_std": 0.057645782828330994, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08772403001785278, "sampling/sampling_logp_difference/max": 1.0531864166259766, "sampling/importance_sampling_ratio/min": 0.34882447123527527, "sampling/importance_sampling_ratio/mean": 1.0191258192062378, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6932972818613052, "clip_ratio/low_mean": 0.010080644860863686, "clip_ratio/low_min": 0.010080644860863686, "clip_ratio/high_mean": 0.06751623563468456, "clip_ratio/high_max": 0.06751623563468456, "clip_ratio/region_mean": 0.07759688049554825, "reward_total_mean": 0.577184796333313, "reward_meter_mean": 0.9075185656547546, "reward_meter_std": 0.2047087401151657, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8628815412521362, "reward_repeat_soft_std": 0.03367576375603676, "reward_judge_quality_mean": 0.41999998688697815, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.577184796333313, "reward_total_composite_std": 0.057645779103040695} {"timestamp_utc": "2026-04-13T12:03:44Z", "mode": "train", "global_step": 1914, "epoch": 0.19226519337016573, "loss": 0.0568, "grad_norm": 23.330875396728516, "learning_rate": 4.203030303030303e-06, "num_tokens": 3441599.0, "completions/mean_length": 48.625, "completions/min_length": 45.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.625, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.618243396282196, "rewards/meter/std": 0.3767756223678589, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9622788429260254, "rewards/repeat_soft/std": 0.0360732264816761, "rewards/judge_quality/mean": 0.8200000524520874, "rewards/judge_quality/std": 0.19272483885288239, "rewards/total_composite/mean": 0.6511150002479553, "rewards/total_composite/std": 0.19084851443767548, "reward": 0.6511150002479553, "reward_std": 0.19084849953651428, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08347024023532867, "sampling/sampling_logp_difference/max": 2.4505133628845215, "sampling/importance_sampling_ratio/min": 0.1143178939819336, "sampling/importance_sampling_ratio/mean": 1.0063972473144531, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2730312719941139, "clip_ratio/low_mean": 0.030320503283292055, "clip_ratio/low_min": 0.030320503283292055, "clip_ratio/high_mean": 0.026296769035980105, "clip_ratio/high_max": 0.026296769035980105, "clip_ratio/region_mean": 0.05661727231927216, "reward_total_mean": 0.6511150002479553, "reward_meter_mean": 0.618243396282196, "reward_meter_std": 0.3767756223678589, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9622788429260254, "reward_repeat_soft_std": 0.0360732264816761, "reward_judge_quality_mean": 0.8200000524520874, "reward_judge_quality_std": 0.19272483885288239, "reward_total_composite_mean": 0.6511150002479553, "reward_total_composite_std": 0.19084851443767548} {"timestamp_utc": "2026-04-13T12:03:53Z", "mode": "train", "global_step": 1915, "epoch": 0.19236564540431944, "loss": -0.0828, "grad_norm": 11.6658353805542, "learning_rate": 4.2000000000000004e-06, "num_tokens": 3444450.0, "completions/mean_length": 159.375, "completions/min_length": 128.0, "completions/max_length": 198.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 159.375, "completions/min_terminated_length": 128.0, "completions/max_terminated_length": 198.0, "rewards/meter/mean": 0.9716150760650635, "rewards/meter/std": 0.01307368092238903, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7041897773742676, "rewards/repeat_soft/std": 0.06158751994371414, "rewards/judge_quality/mean": 0.5950000286102295, "rewards/judge_quality/std": 0.19820626080036163, "rewards/total_composite/mean": 0.6813640594482422, "rewards/total_composite/std": 0.12323393672704697, "reward": 0.6813640594482422, "reward_std": 0.12323394417762756, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07488817721605301, "sampling/sampling_logp_difference/max": 1.7577056884765625, "sampling/importance_sampling_ratio/min": 0.1724400371313095, "sampling/importance_sampling_ratio/mean": 1.002464771270752, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3908664211630821, "clip_ratio/low_mean": 0.03722066944465041, "clip_ratio/low_min": 0.03722066944465041, "clip_ratio/high_mean": 0.03210883028805256, "clip_ratio/high_max": 0.03210883028805256, "clip_ratio/region_mean": 0.06932949973270297, "reward_total_mean": 0.6813640594482422, "reward_meter_mean": 0.9716150760650635, "reward_meter_std": 0.01307368092238903, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7041897773742676, "reward_repeat_soft_std": 0.06158751994371414, "reward_judge_quality_mean": 0.5950000286102295, "reward_judge_quality_std": 0.19820626080036163, "reward_total_composite_mean": 0.6813640594482422, "reward_total_composite_std": 0.12323393672704697} {"timestamp_utc": "2026-04-13T12:04:01Z", "mode": "train", "global_step": 1916, "epoch": 0.19246609743847312, "loss": -0.0325, "grad_norm": 6.624622821807861, "learning_rate": 4.196969696969697e-06, "num_tokens": 3446921.0, "completions/mean_length": 112.875, "completions/min_length": 91.0, "completions/max_length": 131.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.875, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.8507556915283203, "rewards/meter/std": 0.18038310110569, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8237466812133789, "rewards/repeat_soft/std": 0.03414364531636238, "rewards/judge_quality/mean": 0.5774999856948853, "rewards/judge_quality/std": 0.22461079061031342, "rewards/total_composite/mean": 0.6472747325897217, "rewards/total_composite/std": 0.15863634645938873, "reward": 0.6472747325897217, "reward_std": 0.15863634645938873, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.095945343375206, "sampling/sampling_logp_difference/max": 1.421422004699707, "sampling/importance_sampling_ratio/min": 0.2413705438375473, "sampling/importance_sampling_ratio/mean": 1.008117914199829, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4990440383553505, "clip_ratio/low_mean": 0.07097942568361759, "clip_ratio/low_min": 0.07097942568361759, "clip_ratio/high_mean": 0.02008723607286811, "clip_ratio/high_max": 0.02008723607286811, "clip_ratio/region_mean": 0.0910666617564857, "reward_total_mean": 0.6472747325897217, "reward_meter_mean": 0.8507556915283203, "reward_meter_std": 0.18038310110569, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8237466812133789, "reward_repeat_soft_std": 0.03414364531636238, "reward_judge_quality_mean": 0.5774999856948853, "reward_judge_quality_std": 0.22461079061031342, "reward_total_composite_mean": 0.6472747325897217, "reward_total_composite_std": 0.15863634645938873} {"timestamp_utc": "2026-04-13T12:04:07Z", "mode": "train", "global_step": 1917, "epoch": 0.19256654947262683, "loss": -0.0559, "grad_norm": 17.377117156982422, "learning_rate": 4.193939393939394e-06, "num_tokens": 3448367.0, "completions/mean_length": 33.75, "completions/min_length": 26.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.75, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.457536518573761, "rewards/meter/std": 0.35414940118789673, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9779925346374512, "rewards/repeat_soft/std": 0.0339190848171711, "rewards/judge_quality/mean": 0.4650000035762787, "rewards/judge_quality/std": 0.10392305999994278, "rewards/total_composite/mean": 0.47729986906051636, "rewards/total_composite/std": 0.09718886762857437, "reward": 0.47729986906051636, "reward_std": 0.09718887507915497, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14621400833129883, "sampling/sampling_logp_difference/max": 1.2445707321166992, "sampling/importance_sampling_ratio/min": 0.28806453943252563, "sampling/importance_sampling_ratio/mean": 1.0058984756469727, "sampling/importance_sampling_ratio/max": 1.7201218605041504, "entropy": 1.0507483892142773, "clip_ratio/low_mean": 0.03479864727705717, "clip_ratio/low_min": 0.03479864727705717, "clip_ratio/high_mean": 0.05114082945510745, "clip_ratio/high_max": 0.05114082945510745, "clip_ratio/region_mean": 0.08593947673216462, "reward_total_mean": 0.47729986906051636, "reward_meter_mean": 0.457536518573761, "reward_meter_std": 0.35414940118789673, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9779925346374512, "reward_repeat_soft_std": 0.0339190848171711, "reward_judge_quality_mean": 0.4650000035762787, "reward_judge_quality_std": 0.10392305999994278, "reward_total_composite_mean": 0.47729986906051636, "reward_total_composite_std": 0.09718886762857437} {"timestamp_utc": "2026-04-13T12:04:13Z", "mode": "train", "global_step": 1918, "epoch": 0.1926670015067805, "loss": 0.0014, "grad_norm": 18.626001358032227, "learning_rate": 4.190909090909091e-06, "num_tokens": 3449780.0, "completions/mean_length": 26.625, "completions/min_length": 24.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.625, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.8700626492500305, "rewards/meter/std": 0.3448854684829712, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9449080228805542, "rewards/repeat_soft/std": 0.022345291450619698, "rewards/judge_quality/mean": 0.4312499761581421, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.5840761661529541, "rewards/total_composite/std": 0.09695003926753998, "reward": 0.5840761661529541, "reward_std": 0.09695003926753998, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09097051620483398, "sampling/sampling_logp_difference/max": 0.8367805480957031, "sampling/importance_sampling_ratio/min": 0.43310266733169556, "sampling/importance_sampling_ratio/mean": 1.0069423913955688, "sampling/importance_sampling_ratio/max": 1.6343611478805542, "entropy": 0.47952786087989807, "clip_ratio/low_mean": 0.019999999552965164, "clip_ratio/low_min": 0.019999999552965164, "clip_ratio/high_mean": 0.07230510842055082, "clip_ratio/high_max": 0.07230510842055082, "clip_ratio/region_mean": 0.09230510797351599, "reward_total_mean": 0.5840761661529541, "reward_meter_mean": 0.8700626492500305, "reward_meter_std": 0.3448854684829712, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9449080228805542, "reward_repeat_soft_std": 0.022345291450619698, "reward_judge_quality_mean": 0.4312499761581421, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.5840761661529541, "reward_total_composite_std": 0.09695003926753998} {"timestamp_utc": "2026-04-13T12:04:20Z", "mode": "train", "global_step": 1919, "epoch": 0.1927674535409342, "loss": -0.0942, "grad_norm": 7.9500651359558105, "learning_rate": 4.187878787878788e-06, "num_tokens": 3451401.0, "completions/mean_length": 43.625, "completions/min_length": 36.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.625, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.9833869934082031, "rewards/meter/std": 0.004617537371814251, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9158543348312378, "rewards/repeat_soft/std": 0.05400705710053444, "rewards/judge_quality/mean": 0.1537500023841858, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.4356558918952942, "rewards/total_composite/std": 0.01236470602452755, "reward": 0.4356558918952942, "reward_std": 0.012364705093204975, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07290530204772949, "sampling/sampling_logp_difference/max": 1.3190256357192993, "sampling/importance_sampling_ratio/min": 0.2673957049846649, "sampling/importance_sampling_ratio/mean": 1.0044238567352295, "sampling/importance_sampling_ratio/max": 1.6072239875793457, "entropy": 0.4352567121386528, "clip_ratio/low_mean": 0.03689136798493564, "clip_ratio/low_min": 0.03689136798493564, "clip_ratio/high_mean": 0.04079236835241318, "clip_ratio/high_max": 0.04079236835241318, "clip_ratio/region_mean": 0.07768373633734882, "reward_total_mean": 0.4356558918952942, "reward_meter_mean": 0.9833869934082031, "reward_meter_std": 0.004617537371814251, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9158543348312378, "reward_repeat_soft_std": 0.05400705710053444, "reward_judge_quality_mean": 0.1537500023841858, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.4356558918952942, "reward_total_composite_std": 0.01236470602452755} {"timestamp_utc": "2026-04-13T12:04:28Z", "mode": "train", "global_step": 1920, "epoch": 0.1928679055750879, "loss": 0.0382, "grad_norm": 5.887547016143799, "learning_rate": 4.184848484848485e-06, "num_tokens": 3454200.0, "completions/mean_length": 143.875, "completions/min_length": 133.0, "completions/max_length": 160.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 143.875, "completions/min_terminated_length": 133.0, "completions/max_terminated_length": 160.0, "rewards/meter/mean": 0.987758219242096, "rewards/meter/std": 0.005048134829849005, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8466187715530396, "rewards/repeat_soft/std": 0.06207213178277016, "rewards/judge_quality/mean": 0.11999999731779099, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4040379524230957, "rewards/total_composite/std": 0.009467762894928455, "reward": 0.4040379524230957, "reward_std": 0.00946776196360588, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09985199570655823, "sampling/sampling_logp_difference/max": 1.9229249954223633, "sampling/importance_sampling_ratio/min": 0.14617876708507538, "sampling/importance_sampling_ratio/mean": 0.9935129284858704, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5678234249353409, "clip_ratio/low_mean": 0.03725255560129881, "clip_ratio/low_min": 0.03725255560129881, "clip_ratio/high_mean": 0.03946106508374214, "clip_ratio/high_max": 0.03946106508374214, "clip_ratio/region_mean": 0.07671362068504095, "reward_total_mean": 0.4040379524230957, "reward_meter_mean": 0.987758219242096, "reward_meter_std": 0.005048134829849005, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8466187715530396, "reward_repeat_soft_std": 0.06207213178277016, "reward_judge_quality_mean": 0.11999999731779099, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4040379524230957, "reward_total_composite_std": 0.009467762894928455} {"timestamp_utc": "2026-04-13T12:04:39Z", "mode": "train", "global_step": 1921, "epoch": 0.19296835760924158, "loss": -0.0847, "grad_norm": 1.4573488235473633, "learning_rate": 4.181818181818182e-06, "num_tokens": 3455565.0, "completions/mean_length": 84.625, "completions/min_length": 15.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 23.571430206298828, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.6292486190795898, "rewards/meter/std": 0.39336225390434265, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9447200298309326, "rewards/repeat_soft/std": 0.02195359766483307, "rewards/judge_quality/mean": 0.13750000298023224, "rewards/judge_quality/std": 0.0353553406894207, "rewards/total_composite/mean": 0.35830748081207275, "rewards/total_composite/std": 0.14822307229042053, "reward": 0.35830748081207275, "reward_std": 0.14822307229042053, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15679016709327698, "sampling/sampling_logp_difference/max": 1.3619756698608398, "sampling/importance_sampling_ratio/min": 0.25615420937538147, "sampling/importance_sampling_ratio/mean": 1.0062261819839478, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0058701038360596, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.14873274974524975, "clip_ratio/high_max": 0.14873274974524975, "clip_ratio/region_mean": 0.14873274974524975, "reward_total_mean": 0.35830748081207275, "reward_meter_mean": 0.6292486190795898, "reward_meter_std": 0.39336225390434265, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9447200298309326, "reward_repeat_soft_std": 0.02195359766483307, "reward_judge_quality_mean": 0.13750000298023224, "reward_judge_quality_std": 0.0353553406894207, "reward_total_composite_mean": 0.35830748081207275, "reward_total_composite_std": 0.14822307229042053} {"timestamp_utc": "2026-04-13T12:04:50Z", "mode": "train", "global_step": 1922, "epoch": 0.1930688096433953, "loss": -0.1185, "grad_norm": 2.2207748889923096, "learning_rate": 4.1787878787878795e-06, "num_tokens": 3457154.0, "completions/mean_length": 99.625, "completions/min_length": 37.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 40.71428680419922, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.7216843366622925, "rewards/meter/std": 0.35830414295196533, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9028103351593018, "rewards/repeat_soft/std": 0.09586292505264282, "rewards/judge_quality/mean": 0.13750000298023224, "rewards/judge_quality/std": 0.0353553406894207, "rewards/total_composite/mean": 0.3565751016139984, "rewards/total_composite/std": 0.15102575719356537, "reward": 0.3565751016139984, "reward_std": 0.15102575719356537, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13488441705703735, "sampling/sampling_logp_difference/max": 2.1499409675598145, "sampling/importance_sampling_ratio/min": 0.11649102717638016, "sampling/importance_sampling_ratio/mean": 1.0255366563796997, "sampling/importance_sampling_ratio/max": 1.8868846893310547, "entropy": 0.8553041182458401, "clip_ratio/low_mean": 0.013513513840734959, "clip_ratio/low_min": 0.013513513840734959, "clip_ratio/high_mean": 0.07870935089886189, "clip_ratio/high_max": 0.07870935089886189, "clip_ratio/region_mean": 0.09222286473959684, "reward_total_mean": 0.3565751016139984, "reward_meter_mean": 0.7216843366622925, "reward_meter_std": 0.35830414295196533, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9028103351593018, "reward_repeat_soft_std": 0.09586292505264282, "reward_judge_quality_mean": 0.13750000298023224, "reward_judge_quality_std": 0.0353553406894207, "reward_total_composite_mean": 0.3565751016139984, "reward_total_composite_std": 0.15102575719356537} {"timestamp_utc": "2026-04-13T12:04:57Z", "mode": "train", "global_step": 1923, "epoch": 0.19316926167754897, "loss": -0.0912, "grad_norm": 11.489913940429688, "learning_rate": 4.175757575757576e-06, "num_tokens": 3458823.0, "completions/mean_length": 52.625, "completions/min_length": 38.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.625, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.5990158915519714, "rewards/meter/std": 0.3955284059047699, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9780142307281494, "rewards/repeat_soft/std": 0.01682078093290329, "rewards/judge_quality/mean": 0.1537500023841858, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.406122088432312, "rewards/total_composite/std": 0.03803851455450058, "reward": 0.406122088432312, "reward_std": 0.03803851455450058, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13839095830917358, "sampling/sampling_logp_difference/max": 1.4287137985229492, "sampling/importance_sampling_ratio/min": 0.23961691558361053, "sampling/importance_sampling_ratio/mean": 1.0154194831848145, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0459632501006126, "clip_ratio/low_mean": 0.07959857024252415, "clip_ratio/low_min": 0.07959857024252415, "clip_ratio/high_mean": 0.0758630963973701, "clip_ratio/high_max": 0.0758630963973701, "clip_ratio/region_mean": 0.15546166663989425, "reward_total_mean": 0.406122088432312, "reward_meter_mean": 0.5990158915519714, "reward_meter_std": 0.3955284059047699, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9780142307281494, "reward_repeat_soft_std": 0.01682078093290329, "reward_judge_quality_mean": 0.1537500023841858, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.406122088432312, "reward_total_composite_std": 0.03803851455450058} {"timestamp_utc": "2026-04-13T12:05:09Z", "mode": "train", "global_step": 1924, "epoch": 0.19326971371170265, "loss": -0.1086, "grad_norm": 1.5600149631500244, "learning_rate": 4.172727272727273e-06, "num_tokens": 3460324.0, "completions/mean_length": 93.625, "completions/min_length": 29.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 33.85714340209961, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.6899369955062866, "rewards/meter/std": 0.3944839835166931, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.90090012550354, "rewards/repeat_soft/std": 0.13305890560150146, "rewards/judge_quality/mean": 0.13750000298023224, "rewards/judge_quality/std": 0.0353553406894207, "rewards/total_composite/mean": 0.35865187644958496, "rewards/total_composite/std": 0.1468174010515213, "reward": 0.35865187644958496, "reward_std": 0.1468174010515213, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05273240804672241, "sampling/sampling_logp_difference/max": 0.9871878027915955, "sampling/importance_sampling_ratio/min": 0.37262311577796936, "sampling/importance_sampling_ratio/mean": 1.01171875, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.23229003325104713, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.03230845835059881, "clip_ratio/high_max": 0.03230845835059881, "clip_ratio/region_mean": 0.03230845835059881, "reward_total_mean": 0.35865187644958496, "reward_meter_mean": 0.6899369955062866, "reward_meter_std": 0.3944839835166931, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.90090012550354, "reward_repeat_soft_std": 0.13305890560150146, "reward_judge_quality_mean": 0.13750000298023224, "reward_judge_quality_std": 0.0353553406894207, "reward_total_composite_mean": 0.35865187644958496, "reward_total_composite_std": 0.1468174010515213} {"timestamp_utc": "2026-04-13T12:05:15Z", "mode": "train", "global_step": 1925, "epoch": 0.19337016574585636, "loss": 0.0353, "grad_norm": 15.171744346618652, "learning_rate": 4.1696969696969705e-06, "num_tokens": 3461805.0, "completions/mean_length": 39.125, "completions/min_length": 34.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.125, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.9467993974685669, "rewards/meter/std": 0.08223433792591095, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9046837091445923, "rewards/repeat_soft/std": 0.0497312992811203, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.42801550030708313, "rewards/total_composite/std": 0.010744020342826843, "reward": 0.42801550030708313, "reward_std": 0.010744019411504269, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12122054398059845, "sampling/sampling_logp_difference/max": 1.4478511810302734, "sampling/importance_sampling_ratio/min": 0.354824960231781, "sampling/importance_sampling_ratio/mean": 1.0006885528564453, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.939532570540905, "clip_ratio/low_mean": 0.030678459908813238, "clip_ratio/low_min": 0.030678459908813238, "clip_ratio/high_mean": 0.052022610791027546, "clip_ratio/high_max": 0.052022610791027546, "clip_ratio/region_mean": 0.08270107069984078, "reward_total_mean": 0.42801550030708313, "reward_meter_mean": 0.9467993974685669, "reward_meter_std": 0.08223433792591095, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9046837091445923, "reward_repeat_soft_std": 0.0497312992811203, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.42801550030708313, "reward_total_composite_std": 0.010744020342826843} {"timestamp_utc": "2026-04-13T12:05:26Z", "mode": "train", "global_step": 1926, "epoch": 0.19347061778001004, "loss": -0.1535, "grad_norm": 1.6101049184799194, "learning_rate": 4.166666666666667e-06, "num_tokens": 3463584.0, "completions/mean_length": 115.375, "completions/min_length": 50.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 58.71428680419922, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.8866132497787476, "rewards/meter/std": 0.24672026932239532, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9512063264846802, "rewards/repeat_soft/std": 0.038331110030412674, "rewards/judge_quality/mean": 0.14125001430511475, "rewards/judge_quality/std": 0.038335926830768585, "rewards/total_composite/mean": 0.3850158452987671, "rewards/total_composite/std": 0.15585671365261078, "reward": 0.3850158452987671, "reward_std": 0.15585669875144958, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11420281231403351, "sampling/sampling_logp_difference/max": 2.4861817359924316, "sampling/importance_sampling_ratio/min": 0.08322714269161224, "sampling/importance_sampling_ratio/mean": 1.0202020406723022, "sampling/importance_sampling_ratio/max": 1.828543782234192, "entropy": 0.798855647444725, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09341429825872183, "clip_ratio/high_max": 0.09341429825872183, "clip_ratio/region_mean": 0.09341429825872183, "reward_total_mean": 0.3850158452987671, "reward_meter_mean": 0.8866132497787476, "reward_meter_std": 0.24672026932239532, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9512063264846802, "reward_repeat_soft_std": 0.038331110030412674, "reward_judge_quality_mean": 0.14125001430511475, "reward_judge_quality_std": 0.038335926830768585, "reward_total_composite_mean": 0.3850158452987671, "reward_total_composite_std": 0.15585671365261078} {"timestamp_utc": "2026-04-13T12:05:32Z", "mode": "train", "global_step": 1927, "epoch": 0.19357106981416375, "loss": 0.0036, "grad_norm": 11.964699745178223, "learning_rate": 4.163636363636364e-06, "num_tokens": 3465041.0, "completions/mean_length": 23.125, "completions/min_length": 20.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.125, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.994118332862854, "rewards/meter/std": 0.0013978873612359166, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9560492038726807, "rewards/repeat_soft/std": 0.010533427819609642, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.44033390283584595, "rewards/total_composite/std": 0.001578116905875504, "reward": 0.44033390283584595, "reward_std": 0.0015781193505972624, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10783885419368744, "sampling/sampling_logp_difference/max": 0.7501993179321289, "sampling/importance_sampling_ratio/min": 0.4722723960876465, "sampling/importance_sampling_ratio/mean": 1.0312190055847168, "sampling/importance_sampling_ratio/max": 1.8718894720077515, "entropy": 0.8328215926885605, "clip_ratio/low_mean": 0.04190476145595312, "clip_ratio/low_min": 0.04190476145595312, "clip_ratio/high_mean": 0.08828983781859279, "clip_ratio/high_max": 0.08828983781859279, "clip_ratio/region_mean": 0.1301945992745459, "reward_total_mean": 0.44033390283584595, "reward_meter_mean": 0.994118332862854, "reward_meter_std": 0.0013978873612359166, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9560492038726807, "reward_repeat_soft_std": 0.010533427819609642, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.44033390283584595, "reward_total_composite_std": 0.001578116905875504} {"timestamp_utc": "2026-04-13T12:05:38Z", "mode": "train", "global_step": 1928, "epoch": 0.19367152184831743, "loss": 0.0531, "grad_norm": 19.441036224365234, "learning_rate": 4.160606060606061e-06, "num_tokens": 3466389.0, "completions/mean_length": 23.5, "completions/min_length": 20.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.5, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9700110554695129, "rewards/meter/std": 0.028271492570638657, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9606921672821045, "rewards/repeat_soft/std": 0.005113361403346062, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.43867990374565125, "rewards/total_composite/std": 0.003153893630951643, "reward": 0.43867990374565125, "reward_std": 0.003153891069814563, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13899697363376617, "sampling/sampling_logp_difference/max": 1.5593090057373047, "sampling/importance_sampling_ratio/min": 0.21028132736682892, "sampling/importance_sampling_ratio/mean": 1.0141392946243286, "sampling/importance_sampling_ratio/max": 1.9066965579986572, "entropy": 0.9456418007612228, "clip_ratio/low_mean": 0.0297413794323802, "clip_ratio/low_min": 0.0297413794323802, "clip_ratio/high_mean": 0.08298542629927397, "clip_ratio/high_max": 0.08298542629927397, "clip_ratio/region_mean": 0.11272680573165417, "reward_total_mean": 0.43867990374565125, "reward_meter_mean": 0.9700110554695129, "reward_meter_std": 0.028271492570638657, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9606921672821045, "reward_repeat_soft_std": 0.005113361403346062, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.43867990374565125, "reward_total_composite_std": 0.003153893630951643} {"timestamp_utc": "2026-04-13T12:05:44Z", "mode": "train", "global_step": 1929, "epoch": 0.1937719738824711, "loss": 0.0178, "grad_norm": 14.826068878173828, "learning_rate": 4.157575757575758e-06, "num_tokens": 3468025.0, "completions/mean_length": 50.5, "completions/min_length": 34.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.5, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9803265333175659, "rewards/meter/std": 0.021168824285268784, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.88011634349823, "rewards/repeat_soft/std": 0.043118324130773544, "rewards/judge_quality/mean": 0.14625000953674316, "rewards/judge_quality/std": 0.010606604628264904, "rewards/total_composite/mean": 0.4252009093761444, "rewards/total_composite/std": 0.012539559043943882, "reward": 0.4252009093761444, "reward_std": 0.012539563700556755, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11705910414457321, "sampling/sampling_logp_difference/max": 1.5417990684509277, "sampling/importance_sampling_ratio/min": 0.21399575471878052, "sampling/importance_sampling_ratio/mean": 1.0303473472595215, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7311799600720406, "clip_ratio/low_mean": 0.026612902991473675, "clip_ratio/low_min": 0.026612902991473675, "clip_ratio/high_mean": 0.07826747675426304, "clip_ratio/high_max": 0.07826747675426304, "clip_ratio/region_mean": 0.10488037974573672, "reward_total_mean": 0.4252009093761444, "reward_meter_mean": 0.9803265333175659, "reward_meter_std": 0.021168824285268784, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.88011634349823, "reward_repeat_soft_std": 0.043118324130773544, "reward_judge_quality_mean": 0.14625000953674316, "reward_judge_quality_std": 0.010606604628264904, "reward_total_composite_mean": 0.4252009093761444, "reward_total_composite_std": 0.012539559043943882} {"timestamp_utc": "2026-04-13T12:05:58Z", "mode": "train", "global_step": 1930, "epoch": 0.19387242591662482, "loss": 0.0846, "grad_norm": 11.098047256469727, "learning_rate": 4.154545454545455e-06, "num_tokens": 3469790.0, "completions/mean_length": 61.625, "completions/min_length": 44.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.625, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.7586449980735779, "rewards/meter/std": 0.34413576126098633, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9118528366088867, "rewards/repeat_soft/std": 0.034493397921323776, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4107457995414734, "rewards/total_composite/std": 0.03100598230957985, "reward": 0.4107457995414734, "reward_std": 0.031005987897515297, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11681599169969559, "sampling/sampling_logp_difference/max": 1.6404461860656738, "sampling/importance_sampling_ratio/min": 0.19389350712299347, "sampling/importance_sampling_ratio/mean": 1.006695032119751, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5450751818716526, "clip_ratio/low_mean": 0.024801588617265224, "clip_ratio/low_min": 0.024801588617265224, "clip_ratio/high_mean": 0.09530821582302451, "clip_ratio/high_max": 0.09530821582302451, "clip_ratio/region_mean": 0.12010980444028974, "reward_total_mean": 0.4107457995414734, "reward_meter_mean": 0.7586449980735779, "reward_meter_std": 0.34413576126098633, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9118528366088867, "reward_repeat_soft_std": 0.034493397921323776, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4107457995414734, "reward_total_composite_std": 0.03100598230957985} {"timestamp_utc": "2026-04-13T12:06:04Z", "mode": "train", "global_step": 1931, "epoch": 0.1939728779507785, "loss": 0.0894, "grad_norm": 7.9793477058410645, "learning_rate": 4.151515151515152e-06, "num_tokens": 3471453.0, "completions/mean_length": 58.875, "completions/min_length": 41.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.875, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9689643383026123, "rewards/meter/std": 0.022389547899365425, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8962827920913696, "rewards/repeat_soft/std": 0.040072936564683914, "rewards/judge_quality/mean": 0.32500001788139343, "rewards/judge_quality/std": 0.27485060691833496, "rewards/total_composite/mean": 0.5389706492424011, "rewards/total_composite/std": 0.17612947523593903, "reward": 0.5389706492424011, "reward_std": 0.17612949013710022, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09707507491111755, "sampling/sampling_logp_difference/max": 1.8834095001220703, "sampling/importance_sampling_ratio/min": 0.15207073092460632, "sampling/importance_sampling_ratio/mean": 1.0107446908950806, "sampling/importance_sampling_ratio/max": 1.7325270175933838, "entropy": 0.5837307199835777, "clip_ratio/low_mean": 0.05217541428282857, "clip_ratio/low_min": 0.05217541428282857, "clip_ratio/high_mean": 0.03508944879285991, "clip_ratio/high_max": 0.03508944879285991, "clip_ratio/region_mean": 0.08726486307568848, "reward_total_mean": 0.5389706492424011, "reward_meter_mean": 0.9689643383026123, "reward_meter_std": 0.022389547899365425, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8962827920913696, "reward_repeat_soft_std": 0.040072936564683914, "reward_judge_quality_mean": 0.32500001788139343, "reward_judge_quality_std": 0.27485060691833496, "reward_total_composite_mean": 0.5389706492424011, "reward_total_composite_std": 0.17612947523593903} {"timestamp_utc": "2026-04-13T12:06:12Z", "mode": "train", "global_step": 1932, "epoch": 0.1940733299849322, "loss": 0.0345, "grad_norm": 7.466383457183838, "learning_rate": 4.148484848484849e-06, "num_tokens": 3473801.0, "completions/mean_length": 114.5, "completions/min_length": 93.0, "completions/max_length": 143.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 114.5, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 143.0, "rewards/meter/mean": 0.82825767993927, "rewards/meter/std": 0.2862468361854553, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9518787860870361, "rewards/repeat_soft/std": 0.024463597685098648, "rewards/judge_quality/mean": 0.16875001788139343, "rewards/judge_quality/std": 0.022320719435811043, "rewards/total_composite/mean": 0.4324488639831543, "rewards/total_composite/std": 0.0321044884622097, "reward": 0.4324488639831543, "reward_std": 0.0321044959127903, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11769791692495346, "sampling/sampling_logp_difference/max": 1.4505510330200195, "sampling/importance_sampling_ratio/min": 0.23444108664989471, "sampling/importance_sampling_ratio/mean": 1.0082027912139893, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8597783148288727, "clip_ratio/low_mean": 0.04211760591715574, "clip_ratio/low_min": 0.04211760591715574, "clip_ratio/high_mean": 0.0752584207803011, "clip_ratio/high_max": 0.0752584207803011, "clip_ratio/region_mean": 0.11737602669745684, "reward_total_mean": 0.4324488639831543, "reward_meter_mean": 0.82825767993927, "reward_meter_std": 0.2862468361854553, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9518787860870361, "reward_repeat_soft_std": 0.024463597685098648, "reward_judge_quality_mean": 0.16875001788139343, "reward_judge_quality_std": 0.022320719435811043, "reward_total_composite_mean": 0.4324488639831543, "reward_total_composite_std": 0.0321044884622097} {"timestamp_utc": "2026-04-13T12:06:23Z", "mode": "train", "global_step": 1933, "epoch": 0.1941737820190859, "loss": -0.1958, "grad_norm": 1.8619639873504639, "learning_rate": 4.145454545454546e-06, "num_tokens": 3475994.0, "completions/mean_length": 160.125, "completions/min_length": 96.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 109.85714721679688, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 142.0, "rewards/meter/mean": 0.8007317781448364, "rewards/meter/std": 0.31228408217430115, "rewards/count_adherence/mean": 0.925000011920929, "rewards/count_adherence/std": 0.2121320217847824, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8632869720458984, "rewards/repeat_soft/std": 0.08443199098110199, "rewards/judge_quality/mean": 0.15625, "rewards/judge_quality/std": 0.047790467739105225, "rewards/total_composite/mean": 0.402592271566391, "rewards/total_composite/std": 0.07177911698818207, "reward": 0.402592271566391, "reward_std": 0.07177910208702087, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10312580317258835, "sampling/sampling_logp_difference/max": 1.4721665382385254, "sampling/importance_sampling_ratio/min": 0.22942787408828735, "sampling/importance_sampling_ratio/mean": 1.009140968322754, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5213595479726791, "clip_ratio/low_mean": 0.006433823611587286, "clip_ratio/low_min": 0.006433823611587286, "clip_ratio/high_mean": 0.0725303990766406, "clip_ratio/high_max": 0.0725303990766406, "clip_ratio/region_mean": 0.07896422268822789, "reward_total_mean": 0.402592271566391, "reward_meter_mean": 0.8007317781448364, "reward_meter_std": 0.31228408217430115, "reward_count_adherence_mean": 0.925000011920929, "reward_count_adherence_std": 0.2121320217847824, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8632869720458984, "reward_repeat_soft_std": 0.08443199098110199, "reward_judge_quality_mean": 0.15625, "reward_judge_quality_std": 0.047790467739105225, "reward_total_composite_mean": 0.402592271566391, "reward_total_composite_std": 0.07177911698818207} {"timestamp_utc": "2026-04-13T12:06:31Z", "mode": "train", "global_step": 1934, "epoch": 0.19427423405323957, "loss": 0.0111, "grad_norm": 7.697418212890625, "learning_rate": 4.142424242424243e-06, "num_tokens": 3477957.0, "completions/mean_length": 82.375, "completions/min_length": 70.0, "completions/max_length": 93.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.375, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.7867481708526611, "rewards/meter/std": 0.3093531131744385, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9399130344390869, "rewards/repeat_soft/std": 0.030503612011671066, "rewards/judge_quality/mean": 0.21000000834465027, "rewards/judge_quality/std": 0.0975411981344223, "rewards/total_composite/mean": 0.45291727781295776, "rewards/total_composite/std": 0.08042343705892563, "reward": 0.45291727781295776, "reward_std": 0.08042343705892563, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10768397152423859, "sampling/sampling_logp_difference/max": 1.8874082565307617, "sampling/importance_sampling_ratio/min": 0.15146386623382568, "sampling/importance_sampling_ratio/mean": 1.0010343790054321, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6115433350205421, "clip_ratio/low_mean": 0.07161152362823486, "clip_ratio/low_min": 0.07161152362823486, "clip_ratio/high_mean": 0.0290178582072258, "clip_ratio/high_max": 0.0290178582072258, "clip_ratio/region_mean": 0.10062938183546066, "reward_total_mean": 0.45291727781295776, "reward_meter_mean": 0.7867481708526611, "reward_meter_std": 0.3093531131744385, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9399130344390869, "reward_repeat_soft_std": 0.030503612011671066, "reward_judge_quality_mean": 0.21000000834465027, "reward_judge_quality_std": 0.0975411981344223, "reward_total_composite_mean": 0.45291727781295776, "reward_total_composite_std": 0.08042343705892563} {"timestamp_utc": "2026-04-13T12:06:38Z", "mode": "train", "global_step": 1935, "epoch": 0.19437468608739328, "loss": -0.0081, "grad_norm": 9.461777687072754, "learning_rate": 4.13939393939394e-06, "num_tokens": 3479615.0, "completions/mean_length": 57.25, "completions/min_length": 49.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.25, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.7915312647819519, "rewards/meter/std": 0.3439336121082306, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8316245079040527, "rewards/repeat_soft/std": 0.066276915371418, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.401917964220047, "rewards/total_composite/std": 0.032592639327049255, "reward": 0.401917964220047, "reward_std": 0.032592643052339554, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0932314544916153, "sampling/sampling_logp_difference/max": 1.5124073028564453, "sampling/importance_sampling_ratio/min": 0.2203788310289383, "sampling/importance_sampling_ratio/mean": 1.012587308883667, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6603519469499588, "clip_ratio/low_mean": 0.025438006035983562, "clip_ratio/low_min": 0.025438006035983562, "clip_ratio/high_mean": 0.06806070310994983, "clip_ratio/high_max": 0.06806070310994983, "clip_ratio/region_mean": 0.09349870914593339, "reward_total_mean": 0.401917964220047, "reward_meter_mean": 0.7915312647819519, "reward_meter_std": 0.3439336121082306, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8316245079040527, "reward_repeat_soft_std": 0.066276915371418, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.401917964220047, "reward_total_composite_std": 0.032592639327049255} {"timestamp_utc": "2026-04-13T12:06:44Z", "mode": "train", "global_step": 1936, "epoch": 0.19447513812154696, "loss": 0.0203, "grad_norm": 9.021432876586914, "learning_rate": 4.136363636363637e-06, "num_tokens": 3481209.0, "completions/mean_length": 42.25, "completions/min_length": 36.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.25, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.7995381355285645, "rewards/meter/std": 0.18903721868991852, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9265373349189758, "rewards/repeat_soft/std": 0.04161253198981285, "rewards/judge_quality/mean": 0.24625001847743988, "rewards/judge_quality/std": 0.2722361087799072, "rewards/total_composite/mean": 0.42288756370544434, "rewards/total_composite/std": 0.23189647495746613, "reward": 0.42288756370544434, "reward_std": 0.23189647495746613, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12788273394107819, "sampling/sampling_logp_difference/max": 1.9591566324234009, "sampling/importance_sampling_ratio/min": 0.14097726345062256, "sampling/importance_sampling_ratio/mean": 0.9938132762908936, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8986107259988785, "clip_ratio/low_mean": 0.042058270424604416, "clip_ratio/low_min": 0.042058270424604416, "clip_ratio/high_mean": 0.07521497923880816, "clip_ratio/high_max": 0.07521497923880816, "clip_ratio/region_mean": 0.11727324966341257, "reward_total_mean": 0.42288756370544434, "reward_meter_mean": 0.7995381355285645, "reward_meter_std": 0.18903721868991852, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9265373349189758, "reward_repeat_soft_std": 0.04161253198981285, "reward_judge_quality_mean": 0.24625001847743988, "reward_judge_quality_std": 0.2722361087799072, "reward_total_composite_mean": 0.42288756370544434, "reward_total_composite_std": 0.23189647495746613} {"timestamp_utc": "2026-04-13T12:06:51Z", "mode": "train", "global_step": 1937, "epoch": 0.19457559015570064, "loss": 0.0372, "grad_norm": 12.0358247756958, "learning_rate": 4.133333333333333e-06, "num_tokens": 3482851.0, "completions/mean_length": 49.25, "completions/min_length": 39.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.25, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.7147437930107117, "rewards/meter/std": 0.32077938318252563, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8967220783233643, "rewards/repeat_soft/std": 0.0788450762629509, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4041958451271057, "rewards/total_composite/std": 0.03756310045719147, "reward": 0.4041958451271057, "reward_std": 0.03756310045719147, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10153629630804062, "sampling/sampling_logp_difference/max": 1.3987922668457031, "sampling/importance_sampling_ratio/min": 0.246894970536232, "sampling/importance_sampling_ratio/mean": 1.0087206363677979, "sampling/importance_sampling_ratio/max": 1.8293730020523071, "entropy": 0.7433633059263229, "clip_ratio/low_mean": 0.03178792679682374, "clip_ratio/low_min": 0.03178792679682374, "clip_ratio/high_mean": 0.06418938236311078, "clip_ratio/high_max": 0.06418938236311078, "clip_ratio/region_mean": 0.09597730915993452, "reward_total_mean": 0.4041958451271057, "reward_meter_mean": 0.7147437930107117, "reward_meter_std": 0.32077938318252563, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8967220783233643, "reward_repeat_soft_std": 0.0788450762629509, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4041958451271057, "reward_total_composite_std": 0.03756310045719147} {"timestamp_utc": "2026-04-13T12:07:02Z", "mode": "train", "global_step": 1938, "epoch": 0.19467604218985435, "loss": -0.0854, "grad_norm": 1.6866087913513184, "learning_rate": 4.1303030303030305e-06, "num_tokens": 3484174.0, "completions/mean_length": 84.375, "completions/min_length": 18.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 23.285715103149414, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.26195597648620605, "rewards/meter/std": 0.3709193170070648, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9555385112762451, "rewards/repeat_soft/std": 0.019451912492513657, "rewards/judge_quality/mean": 0.13750000298023224, "rewards/judge_quality/std": 0.0353553406894207, "rewards/total_composite/mean": 0.3258320391178131, "rewards/total_composite/std": 0.13631708920001984, "reward": 0.3258320391178131, "reward_std": 0.13631707429885864, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11393966525793076, "sampling/sampling_logp_difference/max": 1.4023149013519287, "sampling/importance_sampling_ratio/min": 0.24602676928043365, "sampling/importance_sampling_ratio/mean": 1.0280745029449463, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6536348387598991, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.10601353459060192, "clip_ratio/high_max": 0.10601353459060192, "clip_ratio/region_mean": 0.10601353459060192, "reward_total_mean": 0.3258320391178131, "reward_meter_mean": 0.26195597648620605, "reward_meter_std": 0.3709193170070648, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9555385112762451, "reward_repeat_soft_std": 0.019451912492513657, "reward_judge_quality_mean": 0.13750000298023224, "reward_judge_quality_std": 0.0353553406894207, "reward_total_composite_mean": 0.3258320391178131, "reward_total_composite_std": 0.13631708920001984} {"timestamp_utc": "2026-04-13T12:07:08Z", "mode": "train", "global_step": 1939, "epoch": 0.19477649422400803, "loss": 0.1062, "grad_norm": 23.812334060668945, "learning_rate": 4.127272727272728e-06, "num_tokens": 3485617.0, "completions/mean_length": 25.375, "completions/min_length": 21.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.375, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9654932618141174, "rewards/meter/std": 0.06134282797574997, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9411230087280273, "rewards/repeat_soft/std": 0.03368011489510536, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.43530404567718506, "rewards/total_composite/std": 0.009656086564064026, "reward": 0.43530404567718506, "reward_std": 0.00965608935803175, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15782931447029114, "sampling/sampling_logp_difference/max": 2.1210851669311523, "sampling/importance_sampling_ratio/min": 0.11990144103765488, "sampling/importance_sampling_ratio/mean": 1.0210922956466675, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0095732435584068, "clip_ratio/low_mean": 0.03354978375136852, "clip_ratio/low_min": 0.03354978375136852, "clip_ratio/high_mean": 0.11721561383455992, "clip_ratio/high_max": 0.11721561383455992, "clip_ratio/region_mean": 0.15076539758592844, "reward_total_mean": 0.43530404567718506, "reward_meter_mean": 0.9654932618141174, "reward_meter_std": 0.06134282797574997, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9411230087280273, "reward_repeat_soft_std": 0.03368011489510536, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.43530404567718506, "reward_total_composite_std": 0.009656086564064026} {"timestamp_utc": "2026-04-13T12:07:21Z", "mode": "train", "global_step": 1940, "epoch": 0.19487694625816174, "loss": -0.1536, "grad_norm": 1.700808048248291, "learning_rate": 4.124242424242424e-06, "num_tokens": 3487447.0, "completions/mean_length": 114.75, "completions/min_length": 46.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 58.000003814697266, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9692917466163635, "rewards/meter/std": 0.04168885573744774, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9311715364456177, "rewards/repeat_soft/std": 0.030505457893013954, "rewards/judge_quality/mean": 0.13750000298023224, "rewards/judge_quality/std": 0.0353553406894207, "rewards/total_composite/mean": 0.3798644542694092, "rewards/total_composite/std": 0.15350237488746643, "reward": 0.3798644542694092, "reward_std": 0.15350237488746643, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1226680725812912, "sampling/sampling_logp_difference/max": 1.3897569179534912, "sampling/importance_sampling_ratio/min": 0.2491358518600464, "sampling/importance_sampling_ratio/mean": 1.0107020139694214, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8404737338423729, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.10910185892134905, "clip_ratio/high_max": 0.10910185892134905, "clip_ratio/region_mean": 0.10910185892134905, "reward_total_mean": 0.3798644542694092, "reward_meter_mean": 0.9692917466163635, "reward_meter_std": 0.04168885573744774, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9311715364456177, "reward_repeat_soft_std": 0.030505457893013954, "reward_judge_quality_mean": 0.13750000298023224, "reward_judge_quality_std": 0.0353553406894207, "reward_total_composite_mean": 0.3798644542694092, "reward_total_composite_std": 0.15350237488746643} {"timestamp_utc": "2026-04-13T12:07:28Z", "mode": "train", "global_step": 1941, "epoch": 0.19497739829231542, "loss": -0.01, "grad_norm": 7.8766961097717285, "learning_rate": 4.1212121212121215e-06, "num_tokens": 3489273.0, "completions/mean_length": 52.25, "completions/min_length": 43.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.25, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.6973185539245605, "rewards/meter/std": 0.2894766330718994, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9258058071136475, "rewards/repeat_soft/std": 0.03156572952866554, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.406859427690506, "rewards/total_composite/std": 0.02906632423400879, "reward": 0.406859427690506, "reward_std": 0.02906632237136364, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08351286500692368, "sampling/sampling_logp_difference/max": 1.7317934036254883, "sampling/importance_sampling_ratio/min": 0.17696675658226013, "sampling/importance_sampling_ratio/mean": 1.0031673908233643, "sampling/importance_sampling_ratio/max": 1.802559494972229, "entropy": 0.5053677409887314, "clip_ratio/low_mean": 0.03865512926131487, "clip_ratio/low_min": 0.03865512926131487, "clip_ratio/high_mean": 0.030179704073816538, "clip_ratio/high_max": 0.030179704073816538, "clip_ratio/region_mean": 0.0688348333351314, "reward_total_mean": 0.406859427690506, "reward_meter_mean": 0.6973185539245605, "reward_meter_std": 0.2894766330718994, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9258058071136475, "reward_repeat_soft_std": 0.03156572952866554, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.406859427690506, "reward_total_composite_std": 0.02906632423400879} {"timestamp_utc": "2026-04-13T12:07:35Z", "mode": "train", "global_step": 1942, "epoch": 0.1950778503264691, "loss": 0.0491, "grad_norm": 16.240028381347656, "learning_rate": 4.118181818181819e-06, "num_tokens": 3490902.0, "completions/mean_length": 29.625, "completions/min_length": 25.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.625, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.8388985395431519, "rewards/meter/std": 0.33876854181289673, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9396454691886902, "rewards/repeat_soft/std": 0.030232569202780724, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4227394461631775, "rewards/total_composite/std": 0.03190704807639122, "reward": 0.4227394461631775, "reward_std": 0.03190704807639122, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08028630167245865, "sampling/sampling_logp_difference/max": 2.378615379333496, "sampling/importance_sampling_ratio/min": 0.09267881512641907, "sampling/importance_sampling_ratio/mean": 1.0173338651657104, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.40816619247198105, "clip_ratio/low_mean": 0.007575757801532745, "clip_ratio/low_min": 0.007575757801532745, "clip_ratio/high_mean": 0.052720814011991024, "clip_ratio/high_max": 0.052720814011991024, "clip_ratio/region_mean": 0.06029657181352377, "reward_total_mean": 0.4227394461631775, "reward_meter_mean": 0.8388985395431519, "reward_meter_std": 0.33876854181289673, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9396454691886902, "reward_repeat_soft_std": 0.030232569202780724, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4227394461631775, "reward_total_composite_std": 0.03190704807639122} {"timestamp_utc": "2026-04-13T12:07:43Z", "mode": "train", "global_step": 1943, "epoch": 0.1951783023606228, "loss": 0.0331, "grad_norm": 6.2539472579956055, "learning_rate": 4.115151515151515e-06, "num_tokens": 3492864.0, "completions/mean_length": 91.25, "completions/min_length": 69.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.25, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.8011467456817627, "rewards/meter/std": 0.2510037422180176, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9019615650177002, "rewards/repeat_soft/std": 0.07098674029111862, "rewards/judge_quality/mean": 0.16500000655651093, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.4202427864074707, "rewards/total_composite/std": 0.028576964512467384, "reward": 0.4202427864074707, "reward_std": 0.028576962649822235, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09535551816225052, "sampling/sampling_logp_difference/max": 1.388596534729004, "sampling/importance_sampling_ratio/min": 0.24942512810230255, "sampling/importance_sampling_ratio/mean": 1.0110878944396973, "sampling/importance_sampling_ratio/max": 1.6919692754745483, "entropy": 0.7175753563642502, "clip_ratio/low_mean": 0.04605818632990122, "clip_ratio/low_min": 0.04605818632990122, "clip_ratio/high_mean": 0.05231527891010046, "clip_ratio/high_max": 0.05231527891010046, "clip_ratio/region_mean": 0.09837346524000168, "reward_total_mean": 0.4202427864074707, "reward_meter_mean": 0.8011467456817627, "reward_meter_std": 0.2510037422180176, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9019615650177002, "reward_repeat_soft_std": 0.07098674029111862, "reward_judge_quality_mean": 0.16500000655651093, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.4202427864074707, "reward_total_composite_std": 0.028576964512467384} {"timestamp_utc": "2026-04-13T12:07:51Z", "mode": "train", "global_step": 1944, "epoch": 0.1952787543947765, "loss": 0.0599, "grad_norm": 6.8244500160217285, "learning_rate": 4.112121212121212e-06, "num_tokens": 3495360.0, "completions/mean_length": 141.0, "completions/min_length": 120.0, "completions/max_length": 176.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 141.0, "completions/min_terminated_length": 120.0, "completions/max_terminated_length": 176.0, "rewards/meter/mean": 0.8854571580886841, "rewards/meter/std": 0.15358369052410126, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9386004209518433, "rewards/repeat_soft/std": 0.02755219116806984, "rewards/judge_quality/mean": 0.16875001788139343, "rewards/judge_quality/std": 0.022320719435811043, "rewards/total_composite/mean": 0.43835583329200745, "rewards/total_composite/std": 0.024016335606575012, "reward": 0.43835583329200745, "reward_std": 0.024016333743929863, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11116247624158859, "sampling/sampling_logp_difference/max": 1.5638446807861328, "sampling/importance_sampling_ratio/min": 0.20932970941066742, "sampling/importance_sampling_ratio/mean": 1.0016816854476929, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4846718981862068, "clip_ratio/low_mean": 0.026041666977107525, "clip_ratio/low_min": 0.026041666977107525, "clip_ratio/high_mean": 0.07818500325083733, "clip_ratio/high_max": 0.07818500325083733, "clip_ratio/region_mean": 0.10422667022794485, "reward_total_mean": 0.43835583329200745, "reward_meter_mean": 0.8854571580886841, "reward_meter_std": 0.15358369052410126, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9386004209518433, "reward_repeat_soft_std": 0.02755219116806984, "reward_judge_quality_mean": 0.16875001788139343, "reward_judge_quality_std": 0.022320719435811043, "reward_total_composite_mean": 0.43835583329200745, "reward_total_composite_std": 0.024016335606575012} {"timestamp_utc": "2026-04-13T12:07:58Z", "mode": "train", "global_step": 1945, "epoch": 0.1953792064289302, "loss": -0.0001, "grad_norm": 10.8930082321167, "learning_rate": 4.10909090909091e-06, "num_tokens": 3497337.0, "completions/mean_length": 84.125, "completions/min_length": 76.0, "completions/max_length": 95.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.125, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.5784173011779785, "rewards/meter/std": 0.40984830260276794, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9831427335739136, "rewards/repeat_soft/std": 0.017384203150868416, "rewards/judge_quality/mean": 0.21000000834465027, "rewards/judge_quality/std": 0.0975411981344223, "rewards/total_composite/mean": 0.41480445861816406, "rewards/total_composite/std": 0.0438450463116169, "reward": 0.41480445861816406, "reward_std": 0.043845050036907196, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14168818295001984, "sampling/sampling_logp_difference/max": 2.2974629402160645, "sampling/importance_sampling_ratio/min": 0.1005135253071785, "sampling/importance_sampling_ratio/mean": 0.9871100187301636, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4833308830857277, "clip_ratio/low_mean": 0.04418520908802748, "clip_ratio/low_min": 0.04418520908802748, "clip_ratio/high_mean": 0.068550162948668, "clip_ratio/high_max": 0.068550162948668, "clip_ratio/region_mean": 0.11273537203669548, "reward_total_mean": 0.41480445861816406, "reward_meter_mean": 0.5784173011779785, "reward_meter_std": 0.40984830260276794, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9831427335739136, "reward_repeat_soft_std": 0.017384203150868416, "reward_judge_quality_mean": 0.21000000834465027, "reward_judge_quality_std": 0.0975411981344223, "reward_total_composite_mean": 0.41480445861816406, "reward_total_composite_std": 0.0438450463116169} {"timestamp_utc": "2026-04-13T12:08:04Z", "mode": "train", "global_step": 1946, "epoch": 0.19547965846308388, "loss": 0.0256, "grad_norm": 9.013469696044922, "learning_rate": 4.106060606060606e-06, "num_tokens": 3499330.0, "completions/mean_length": 67.125, "completions/min_length": 56.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.125, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.5188466310501099, "rewards/meter/std": 0.29012858867645264, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9260605573654175, "rewards/repeat_soft/std": 0.05419487506151199, "rewards/judge_quality/mean": 0.1574999988079071, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.3905407786369324, "rewards/total_composite/std": 0.029915479943156242, "reward": 0.3905407786369324, "reward_std": 0.02991548366844654, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1178271472454071, "sampling/sampling_logp_difference/max": 3.939318895339966, "sampling/importance_sampling_ratio/min": 0.019461465999484062, "sampling/importance_sampling_ratio/mean": 0.9934802651405334, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5234301798045635, "clip_ratio/low_mean": 0.061407689936459064, "clip_ratio/low_min": 0.061407689936459064, "clip_ratio/high_mean": 0.045002466067671776, "clip_ratio/high_max": 0.045002466067671776, "clip_ratio/region_mean": 0.10641015600413084, "reward_total_mean": 0.3905407786369324, "reward_meter_mean": 0.5188466310501099, "reward_meter_std": 0.29012858867645264, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9260605573654175, "reward_repeat_soft_std": 0.05419487506151199, "reward_judge_quality_mean": 0.1574999988079071, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.3905407786369324, "reward_total_composite_std": 0.029915479943156242} {"timestamp_utc": "2026-04-13T12:08:11Z", "mode": "train", "global_step": 1947, "epoch": 0.19558011049723756, "loss": 0.139, "grad_norm": 13.41064167022705, "learning_rate": 4.103030303030303e-06, "num_tokens": 3501115.0, "completions/mean_length": 49.125, "completions/min_length": 41.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.125, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.7810739278793335, "rewards/meter/std": 0.32668501138687134, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9738366603851318, "rewards/repeat_soft/std": 0.03120327740907669, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.42223018407821655, "rewards/total_composite/std": 0.030131608247756958, "reward": 0.42223018407821655, "reward_std": 0.030131608247756958, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14453963935375214, "sampling/sampling_logp_difference/max": 1.6577138900756836, "sampling/importance_sampling_ratio/min": 0.19057415425777435, "sampling/importance_sampling_ratio/mean": 1.0178121328353882, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0398504585027695, "clip_ratio/low_mean": 0.05733144469559193, "clip_ratio/low_min": 0.05733144469559193, "clip_ratio/high_mean": 0.11275254096835852, "clip_ratio/high_max": 0.11275254096835852, "clip_ratio/region_mean": 0.17008398566395044, "reward_total_mean": 0.42223018407821655, "reward_meter_mean": 0.7810739278793335, "reward_meter_std": 0.32668501138687134, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9738366603851318, "reward_repeat_soft_std": 0.03120327740907669, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.42223018407821655, "reward_total_composite_std": 0.030131608247756958} {"timestamp_utc": "2026-04-13T12:08:19Z", "mode": "train", "global_step": 1948, "epoch": 0.19568056253139127, "loss": 0.0731, "grad_norm": 12.654583930969238, "learning_rate": 4.1e-06, "num_tokens": 3502794.0, "completions/mean_length": 46.875, "completions/min_length": 39.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.875, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.7894902229309082, "rewards/meter/std": 0.3574954867362976, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9605759978294373, "rewards/repeat_soft/std": 0.03400946035981178, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4210616946220398, "rewards/total_composite/std": 0.03653131425380707, "reward": 0.4210616946220398, "reward_std": 0.03653131425380707, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11745037883520126, "sampling/sampling_logp_difference/max": 1.2075090408325195, "sampling/importance_sampling_ratio/min": 0.2989410161972046, "sampling/importance_sampling_ratio/mean": 0.9976826310157776, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.599471140652895, "clip_ratio/low_mean": 0.023413733579218388, "clip_ratio/low_min": 0.023413733579218388, "clip_ratio/high_mean": 0.09726464096456766, "clip_ratio/high_max": 0.09726464096456766, "clip_ratio/region_mean": 0.12067837454378605, "reward_total_mean": 0.4210616946220398, "reward_meter_mean": 0.7894902229309082, "reward_meter_std": 0.3574954867362976, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9605759978294373, "reward_repeat_soft_std": 0.03400946035981178, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4210616946220398, "reward_total_composite_std": 0.03653131425380707} {"timestamp_utc": "2026-04-13T12:08:25Z", "mode": "train", "global_step": 1949, "epoch": 0.19578101456554495, "loss": -0.0375, "grad_norm": 10.614028930664062, "learning_rate": 4.096969696969697e-06, "num_tokens": 3504563.0, "completions/mean_length": 46.125, "completions/min_length": 38.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.125, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.853850781917572, "rewards/meter/std": 0.20310230553150177, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9574228525161743, "rewards/repeat_soft/std": 0.015529298223555088, "rewards/judge_quality/mean": 0.16500000655651093, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.4341689348220825, "rewards/total_composite/std": 0.018940472975373268, "reward": 0.4341689348220825, "reward_std": 0.018940474838018417, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08881442248821259, "sampling/sampling_logp_difference/max": 1.837799072265625, "sampling/importance_sampling_ratio/min": 0.1591673642396927, "sampling/importance_sampling_ratio/mean": 0.9945177435874939, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4520632103085518, "clip_ratio/low_mean": 0.024074074113741517, "clip_ratio/low_min": 0.024074074113741517, "clip_ratio/high_mean": 0.06723985262215137, "clip_ratio/high_max": 0.06723985262215137, "clip_ratio/region_mean": 0.09131392673589289, "reward_total_mean": 0.4341689348220825, "reward_meter_mean": 0.853850781917572, "reward_meter_std": 0.20310230553150177, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9574228525161743, "reward_repeat_soft_std": 0.015529298223555088, "reward_judge_quality_mean": 0.16500000655651093, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.4341689348220825, "reward_total_composite_std": 0.018940472975373268} {"timestamp_utc": "2026-04-13T12:08:32Z", "mode": "train", "global_step": 1950, "epoch": 0.19588146659969866, "loss": -0.0324, "grad_norm": 6.639441013336182, "learning_rate": 4.093939393939394e-06, "num_tokens": 3506541.0, "completions/mean_length": 77.25, "completions/min_length": 61.0, "completions/max_length": 95.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.25, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.856492817401886, "rewards/meter/std": 0.23869705200195312, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.86268150806427, "rewards/repeat_soft/std": 0.03397032245993614, "rewards/judge_quality/mean": 0.23625001311302185, "rewards/judge_quality/std": 0.13265825808048248, "rewards/total_composite/mean": 0.46340858936309814, "rewards/total_composite/std": 0.08909754455089569, "reward": 0.46340858936309814, "reward_std": 0.0890975371003151, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08762527257204056, "sampling/sampling_logp_difference/max": 2.2014412879943848, "sampling/importance_sampling_ratio/min": 0.11064357310533524, "sampling/importance_sampling_ratio/mean": 1.025205373764038, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4552963748574257, "clip_ratio/low_mean": 0.06273894617334008, "clip_ratio/low_min": 0.06273894617334008, "clip_ratio/high_mean": 0.0182926831766963, "clip_ratio/high_max": 0.0182926831766963, "clip_ratio/region_mean": 0.08103162935003638, "reward_total_mean": 0.46340858936309814, "reward_meter_mean": 0.856492817401886, "reward_meter_std": 0.23869705200195312, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.86268150806427, "reward_repeat_soft_std": 0.03397032245993614, "reward_judge_quality_mean": 0.23625001311302185, "reward_judge_quality_std": 0.13265825808048248, "reward_total_composite_mean": 0.46340858936309814, "reward_total_composite_std": 0.08909754455089569} {"timestamp_utc": "2026-04-13T12:09:20Z", "mode": "eval", "global_step": 1950, "epoch": 0.19588146659969866, "eval_loss": NaN, "eval_runtime": 47.3851, "eval_samples_per_second": 1.688, "eval_steps_per_second": 0.211, "eval_num_tokens": 3506541.0, "eval_completions/mean_length": 87.3375, "eval_completions/min_length": 34.0, "eval_completions/max_length": 179.1, "eval_completions/clipped_ratio": 0.0125, "eval_completions/mean_terminated_length": 82.48571472167968, "eval_completions/min_terminated_length": 34.0, "eval_completions/max_terminated_length": 145.7, "eval_rewards/meter/mean": 0.8456712782382965, "eval_rewards/meter/std": 0.25767267104238273, "eval_rewards/count_adherence/mean": 0.990625, "eval_rewards/count_adherence/std": 0.02651650384068489, "eval_rewards/hard_gate/mean": 0.9875, "eval_rewards/hard_gate/std": 0.03535533845424652, "eval_rewards/repeat_soft/mean": 0.8660417795181274, "eval_rewards/repeat_soft/std": 0.10166475251317024, "eval_rewards/judge_quality/mean": 0.1618750050663948, "eval_rewards/judge_quality/std": 0.028590949438512325, "eval_rewards/total_composite/mean": 0.41445557177066805, "eval_rewards/total_composite/std": 0.04768334738910198, "eval_reward": 0.41445557177066805, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.04655205011367798, "eval_sampling/sampling_logp_difference/max": 0.9743308305740357, "eval_sampling/importance_sampling_ratio/min": 0.38960790634155273, "eval_sampling/importance_sampling_ratio/mean": 1.0100993633270263, "eval_sampling/importance_sampling_ratio/max": 1.427927815914154, "eval_entropy": 0.5144751250743866, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.41445557177066805, "eval_reward_meter_mean": 0.8456712782382965, "eval_reward_meter_std": 0.25767267104238273, "eval_reward_count_adherence_mean": 0.990625, "eval_reward_count_adherence_std": 0.02651650384068489, "eval_reward_hard_gate_mean": 0.9875, "eval_reward_hard_gate_std": 0.03535533845424652, "eval_reward_repeat_soft_mean": 0.8660417795181274, "eval_reward_repeat_soft_std": 0.10166475251317024, "eval_reward_judge_quality_mean": 0.1618750050663948, "eval_reward_judge_quality_std": 0.028590949438512325, "eval_reward_total_composite_mean": 0.41445557177066805, "eval_reward_total_composite_std": 0.04768334738910198} {"timestamp_utc": "2026-04-13T12:09:30Z", "mode": "train", "global_step": 1951, "epoch": 0.19598191863385234, "loss": -0.038, "grad_norm": 10.133533477783203, "learning_rate": 4.0909090909090915e-06, "num_tokens": 3508186.0, "completions/mean_length": 52.625, "completions/min_length": 43.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.625, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.9263145923614502, "rewards/meter/std": 0.11202561110258102, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9856852889060974, "rewards/repeat_soft/std": 0.017451290041208267, "rewards/judge_quality/mean": 0.16124999523162842, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.4454079866409302, "rewards/total_composite/std": 0.016052085906267166, "reward": 0.4454079866409302, "reward_std": 0.016052084043622017, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11850112676620483, "sampling/sampling_logp_difference/max": 1.7425165176391602, "sampling/importance_sampling_ratio/min": 0.17507925629615784, "sampling/importance_sampling_ratio/mean": 0.9920020699501038, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7089521065354347, "clip_ratio/low_mean": 0.07087081763893366, "clip_ratio/low_min": 0.07087081763893366, "clip_ratio/high_mean": 0.05990118533372879, "clip_ratio/high_max": 0.05990118533372879, "clip_ratio/region_mean": 0.13077200297266245, "reward_total_mean": 0.4454079866409302, "reward_meter_mean": 0.9263145923614502, "reward_meter_std": 0.11202561110258102, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9856852889060974, "reward_repeat_soft_std": 0.017451290041208267, "reward_judge_quality_mean": 0.16124999523162842, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.4454079866409302, "reward_total_composite_std": 0.016052085906267166} {"timestamp_utc": "2026-04-13T12:09:36Z", "mode": "train", "global_step": 1952, "epoch": 0.19608237066800602, "loss": 0.0555, "grad_norm": 34.25065994262695, "learning_rate": 4.087878787878789e-06, "num_tokens": 3509693.0, "completions/mean_length": 34.375, "completions/min_length": 28.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.375, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.4445779025554657, "rewards/meter/std": 0.27867090702056885, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9061572551727295, "rewards/repeat_soft/std": 0.013661096803843975, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.37926995754241943, "rewards/total_composite/std": 0.028161024674773216, "reward": 0.37926995754241943, "reward_std": 0.028161028400063515, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10101846605539322, "sampling/sampling_logp_difference/max": 1.9981274604797363, "sampling/importance_sampling_ratio/min": 0.13558894395828247, "sampling/importance_sampling_ratio/mean": 0.9924972057342529, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.39059606939554214, "clip_ratio/low_mean": 0.053744684904813766, "clip_ratio/low_min": 0.053744684904813766, "clip_ratio/high_mean": 0.06362083368003368, "clip_ratio/high_max": 0.06362083368003368, "clip_ratio/region_mean": 0.11736551858484745, "reward_total_mean": 0.37926995754241943, "reward_meter_mean": 0.4445779025554657, "reward_meter_std": 0.27867090702056885, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9061572551727295, "reward_repeat_soft_std": 0.013661096803843975, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.37926995754241943, "reward_total_composite_std": 0.028161024674773216} {"timestamp_utc": "2026-04-13T12:09:43Z", "mode": "train", "global_step": 1953, "epoch": 0.19618282270215973, "loss": 0.0071, "grad_norm": 8.840203285217285, "learning_rate": 4.084848484848485e-06, "num_tokens": 3511276.0, "completions/mean_length": 41.875, "completions/min_length": 36.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.875, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.990298867225647, "rewards/meter/std": 0.00986767839640379, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.899217426776886, "rewards/repeat_soft/std": 0.03236125409603119, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4314367473125458, "rewards/total_composite/std": 0.005005829967558384, "reward": 0.4314367473125458, "reward_std": 0.005005823448300362, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09813302010297775, "sampling/sampling_logp_difference/max": 1.5159664154052734, "sampling/importance_sampling_ratio/min": 0.21959586441516876, "sampling/importance_sampling_ratio/mean": 1.0026865005493164, "sampling/importance_sampling_ratio/max": 1.9148515462875366, "entropy": 0.546804204583168, "clip_ratio/low_mean": 0.0374533380381763, "clip_ratio/low_min": 0.0374533380381763, "clip_ratio/high_mean": 0.06395576894283295, "clip_ratio/high_max": 0.06395576894283295, "clip_ratio/region_mean": 0.10140910698100924, "reward_total_mean": 0.4314367473125458, "reward_meter_mean": 0.990298867225647, "reward_meter_std": 0.00986767839640379, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.899217426776886, "reward_repeat_soft_std": 0.03236125409603119, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4314367473125458, "reward_total_composite_std": 0.005005829967558384} {"timestamp_utc": "2026-04-13T12:09:49Z", "mode": "train", "global_step": 1954, "epoch": 0.1962832747363134, "loss": 0.0691, "grad_norm": 14.853337287902832, "learning_rate": 4.081818181818182e-06, "num_tokens": 3513076.0, "completions/mean_length": 54.0, "completions/min_length": 46.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9747487306594849, "rewards/meter/std": 0.01338912732899189, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.953041136264801, "rewards/repeat_soft/std": 0.03769413009285927, "rewards/judge_quality/mean": 0.14250001311302185, "rewards/judge_quality/std": 0.013887306675314903, "rewards/total_composite/mean": 0.43320900201797485, "rewards/total_composite/std": 0.011402982287108898, "reward": 0.43320900201797485, "reward_std": 0.011402991600334644, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11398535221815109, "sampling/sampling_logp_difference/max": 2.2634103298187256, "sampling/importance_sampling_ratio/min": 0.10399521887302399, "sampling/importance_sampling_ratio/mean": 1.017561912536621, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6152001395821571, "clip_ratio/low_mean": 0.02483258955180645, "clip_ratio/low_min": 0.02483258955180645, "clip_ratio/high_mean": 0.06634244974702597, "clip_ratio/high_max": 0.06634244974702597, "clip_ratio/region_mean": 0.09117503929883242, "reward_total_mean": 0.43320900201797485, "reward_meter_mean": 0.9747487306594849, "reward_meter_std": 0.01338912732899189, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.953041136264801, "reward_repeat_soft_std": 0.03769413009285927, "reward_judge_quality_mean": 0.14250001311302185, "reward_judge_quality_std": 0.013887306675314903, "reward_total_composite_mean": 0.43320900201797485, "reward_total_composite_std": 0.011402982287108898} {"timestamp_utc": "2026-04-13T12:09:56Z", "mode": "train", "global_step": 1955, "epoch": 0.19638372677046712, "loss": 0.0469, "grad_norm": 11.980443000793457, "learning_rate": 4.07878787878788e-06, "num_tokens": 3514807.0, "completions/mean_length": 56.375, "completions/min_length": 48.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.375, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.84230637550354, "rewards/meter/std": 0.32354119420051575, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9703394174575806, "rewards/repeat_soft/std": 0.024043342098593712, "rewards/judge_quality/mean": 0.1875, "rewards/judge_quality/std": 0.1060660108923912, "rewards/total_composite/mean": 0.4507935643196106, "rewards/total_composite/std": 0.07669742405414581, "reward": 0.4507935643196106, "reward_std": 0.07669741660356522, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13434912264347076, "sampling/sampling_logp_difference/max": 1.508545160293579, "sampling/importance_sampling_ratio/min": 0.221231609582901, "sampling/importance_sampling_ratio/mean": 1.013060450553894, "sampling/importance_sampling_ratio/max": 1.9136455059051514, "entropy": 0.8411755263805389, "clip_ratio/low_mean": 0.09615141851827502, "clip_ratio/low_min": 0.09615141851827502, "clip_ratio/high_mean": 0.015625, "clip_ratio/high_max": 0.015625, "clip_ratio/region_mean": 0.11177641851827502, "reward_total_mean": 0.4507935643196106, "reward_meter_mean": 0.84230637550354, "reward_meter_std": 0.32354119420051575, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9703394174575806, "reward_repeat_soft_std": 0.024043342098593712, "reward_judge_quality_mean": 0.1875, "reward_judge_quality_std": 0.1060660108923912, "reward_total_composite_mean": 0.4507935643196106, "reward_total_composite_std": 0.07669742405414581} {"timestamp_utc": "2026-04-13T12:10:02Z", "mode": "train", "global_step": 1956, "epoch": 0.1964841788046208, "loss": -0.0532, "grad_norm": 15.185262680053711, "learning_rate": 4.075757575757576e-06, "num_tokens": 3516375.0, "completions/mean_length": 50.0, "completions/min_length": 38.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.0, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.950300931930542, "rewards/meter/std": 0.053850725293159485, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9755334854125977, "rewards/repeat_soft/std": 0.022816307842731476, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.43898436427116394, "rewards/total_composite/std": 0.005826047156006098, "reward": 0.43898436427116394, "reward_std": 0.0058260527439415455, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1201016753911972, "sampling/sampling_logp_difference/max": 1.6800508499145508, "sampling/importance_sampling_ratio/min": 0.18636450171470642, "sampling/importance_sampling_ratio/mean": 1.0156158208847046, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8416338339447975, "clip_ratio/low_mean": 0.04278420936316252, "clip_ratio/low_min": 0.04278420936316252, "clip_ratio/high_mean": 0.0748578580096364, "clip_ratio/high_max": 0.0748578580096364, "clip_ratio/region_mean": 0.11764206737279892, "reward_total_mean": 0.43898436427116394, "reward_meter_mean": 0.950300931930542, "reward_meter_std": 0.053850725293159485, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9755334854125977, "reward_repeat_soft_std": 0.022816307842731476, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.43898436427116394, "reward_total_composite_std": 0.005826047156006098} {"timestamp_utc": "2026-04-13T12:10:10Z", "mode": "train", "global_step": 1957, "epoch": 0.19658463083877448, "loss": -0.0039, "grad_norm": 4.864860534667969, "learning_rate": 4.072727272727273e-06, "num_tokens": 3519219.0, "completions/mean_length": 159.5, "completions/min_length": 84.0, "completions/max_length": 189.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 159.5, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 189.0, "rewards/meter/mean": 0.8495699167251587, "rewards/meter/std": 0.2617471218109131, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.7231045961380005, "rewards/repeat_soft/std": 0.10974279046058655, "rewards/judge_quality/mean": 0.14250001311302185, "rewards/judge_quality/std": 0.031052954494953156, "rewards/total_composite/mean": 0.3280775547027588, "rewards/total_composite/std": 0.14272281527519226, "reward": 0.3280775547027588, "reward_std": 0.14272280037403107, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07233616709709167, "sampling/sampling_logp_difference/max": 1.684156894683838, "sampling/importance_sampling_ratio/min": 0.18560084700584412, "sampling/importance_sampling_ratio/mean": 1.010395884513855, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.45209553465247154, "clip_ratio/low_mean": 0.016155288321897388, "clip_ratio/low_min": 0.016155288321897388, "clip_ratio/high_mean": 0.05875499313697219, "clip_ratio/high_max": 0.05875499313697219, "clip_ratio/region_mean": 0.07491028145886958, "reward_total_mean": 0.3280775547027588, "reward_meter_mean": 0.8495699167251587, "reward_meter_std": 0.2617471218109131, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.7231045961380005, "reward_repeat_soft_std": 0.10974279046058655, "reward_judge_quality_mean": 0.14250001311302185, "reward_judge_quality_std": 0.031052954494953156, "reward_total_composite_mean": 0.3280775547027588, "reward_total_composite_std": 0.14272281527519226} {"timestamp_utc": "2026-04-13T12:10:16Z", "mode": "train", "global_step": 1958, "epoch": 0.19668508287292819, "loss": 0.0012, "grad_norm": 13.044492721557617, "learning_rate": 4.0696969696969706e-06, "num_tokens": 3520729.0, "completions/mean_length": 34.75, "completions/min_length": 31.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.75, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.8093085289001465, "rewards/meter/std": 0.3347797989845276, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9042980074882507, "rewards/repeat_soft/std": 0.0484519861638546, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4145523011684418, "rewards/total_composite/std": 0.03301135078072548, "reward": 0.4145523011684418, "reward_std": 0.03301134705543518, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09243471175432205, "sampling/sampling_logp_difference/max": 2.3862814903259277, "sampling/importance_sampling_ratio/min": 0.09197104722261429, "sampling/importance_sampling_ratio/mean": 0.9922285676002502, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.46040232852101326, "clip_ratio/low_mean": 0.018939394503831863, "clip_ratio/low_min": 0.018939394503831863, "clip_ratio/high_mean": 0.0745106372050941, "clip_ratio/high_max": 0.0745106372050941, "clip_ratio/region_mean": 0.09345003170892596, "reward_total_mean": 0.4145523011684418, "reward_meter_mean": 0.8093085289001465, "reward_meter_std": 0.3347797989845276, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9042980074882507, "reward_repeat_soft_std": 0.0484519861638546, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4145523011684418, "reward_total_composite_std": 0.03301135078072548} {"timestamp_utc": "2026-04-13T12:10:23Z", "mode": "train", "global_step": 1959, "epoch": 0.19678553490708187, "loss": -0.0112, "grad_norm": 19.16962432861328, "learning_rate": 4.066666666666667e-06, "num_tokens": 3522277.0, "completions/mean_length": 36.5, "completions/min_length": 33.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.5, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.8629001379013062, "rewards/meter/std": 0.27387601137161255, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9339948892593384, "rewards/repeat_soft/std": 0.0631670132279396, "rewards/judge_quality/mean": 0.16500000655651093, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.43170130252838135, "rewards/total_composite/std": 0.027718383818864822, "reward": 0.43170130252838135, "reward_std": 0.02771839313209057, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08232957869768143, "sampling/sampling_logp_difference/max": 2.5419797897338867, "sampling/importance_sampling_ratio/min": 0.0787104144692421, "sampling/importance_sampling_ratio/mean": 0.9993715882301331, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.34769029915332794, "clip_ratio/low_mean": 0.018506493885070086, "clip_ratio/low_min": 0.018506493885070086, "clip_ratio/high_mean": 0.049481700640171766, "clip_ratio/high_max": 0.049481700640171766, "clip_ratio/region_mean": 0.06798819452524185, "reward_total_mean": 0.43170130252838135, "reward_meter_mean": 0.8629001379013062, "reward_meter_std": 0.27387601137161255, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9339948892593384, "reward_repeat_soft_std": 0.0631670132279396, "reward_judge_quality_mean": 0.16500000655651093, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.43170130252838135, "reward_total_composite_std": 0.027718383818864822} {"timestamp_utc": "2026-04-13T12:10:29Z", "mode": "train", "global_step": 1960, "epoch": 0.19688598694123555, "loss": -0.0229, "grad_norm": 18.765748977661133, "learning_rate": 4.063636363636364e-06, "num_tokens": 3523659.0, "completions/mean_length": 24.75, "completions/min_length": 18.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.75, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.6937103271484375, "rewards/meter/std": 0.4257435202598572, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9576764702796936, "rewards/repeat_soft/std": 0.013642913661897182, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.41128823161125183, "rewards/total_composite/std": 0.04105198010802269, "reward": 0.41128823161125183, "reward_std": 0.04105199873447418, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16380755603313446, "sampling/sampling_logp_difference/max": 2.570831775665283, "sampling/importance_sampling_ratio/min": 0.07647190988063812, "sampling/importance_sampling_ratio/mean": 1.0118906497955322, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.050104521214962, "clip_ratio/low_mean": 0.06570512987673283, "clip_ratio/low_min": 0.06570512987673283, "clip_ratio/high_mean": 0.09601280698552728, "clip_ratio/high_max": 0.09601280698552728, "clip_ratio/region_mean": 0.1617179368622601, "reward_total_mean": 0.41128823161125183, "reward_meter_mean": 0.6937103271484375, "reward_meter_std": 0.4257435202598572, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9576764702796936, "reward_repeat_soft_std": 0.013642913661897182, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.41128823161125183, "reward_total_composite_std": 0.04105198010802269} {"timestamp_utc": "2026-04-13T12:10:35Z", "mode": "train", "global_step": 1961, "epoch": 0.19698643897538926, "loss": 0.0007, "grad_norm": 29.277746200561523, "learning_rate": 4.060606060606061e-06, "num_tokens": 3524995.0, "completions/mean_length": 16.0, "completions/min_length": 15.0, "completions/max_length": 18.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 16.0, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 18.0, "rewards/meter/mean": 0.7733647227287292, "rewards/meter/std": 0.34330764412879944, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.24625001847743988, "rewards/judge_quality/std": 0.2722361087799072, "rewards/total_composite/mean": 0.4203203320503235, "rewards/total_composite/std": 0.032099027186632156, "reward": 0.4203203320503235, "reward_std": 0.03209902346134186, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11788924038410187, "sampling/sampling_logp_difference/max": 1.3102974891662598, "sampling/importance_sampling_ratio/min": 0.2697398066520691, "sampling/importance_sampling_ratio/mean": 1.0119284391403198, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6030541844666004, "clip_ratio/low_mean": 0.023958333767950535, "clip_ratio/low_min": 0.023958333767950535, "clip_ratio/high_mean": 0.09982638899236917, "clip_ratio/high_max": 0.09982638899236917, "clip_ratio/region_mean": 0.12378472276031971, "reward_total_mean": 0.4203203320503235, "reward_meter_mean": 0.7733647227287292, "reward_meter_std": 0.34330764412879944, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.24625001847743988, "reward_judge_quality_std": 0.2722361087799072, "reward_total_composite_mean": 0.4203203320503235, "reward_total_composite_std": 0.032099027186632156} {"timestamp_utc": "2026-04-13T12:10:42Z", "mode": "train", "global_step": 1962, "epoch": 0.19708689100954294, "loss": 0.0365, "grad_norm": 8.924705505371094, "learning_rate": 4.057575757575758e-06, "num_tokens": 3526657.0, "completions/mean_length": 44.75, "completions/min_length": 31.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.75, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.9242417216300964, "rewards/meter/std": 0.13994993269443512, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8894854784011841, "rewards/repeat_soft/std": 0.014142140746116638, "rewards/judge_quality/mean": 0.1537500023841858, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.4259297251701355, "rewards/total_composite/std": 0.015334603376686573, "reward": 0.4259297251701355, "reward_std": 0.015334603376686573, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09322226792573929, "sampling/sampling_logp_difference/max": 1.4590911865234375, "sampling/importance_sampling_ratio/min": 0.23244743049144745, "sampling/importance_sampling_ratio/mean": 0.9982884526252747, "sampling/importance_sampling_ratio/max": 1.669023871421814, "entropy": 0.5456208884716034, "clip_ratio/low_mean": 0.010130719048902392, "clip_ratio/low_min": 0.010130719048902392, "clip_ratio/high_mean": 0.08445008657872677, "clip_ratio/high_max": 0.08445008657872677, "clip_ratio/region_mean": 0.09458080562762916, "reward_total_mean": 0.4259297251701355, "reward_meter_mean": 0.9242417216300964, "reward_meter_std": 0.13994993269443512, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8894854784011841, "reward_repeat_soft_std": 0.014142140746116638, "reward_judge_quality_mean": 0.1537500023841858, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.4259297251701355, "reward_total_composite_std": 0.015334603376686573} {"timestamp_utc": "2026-04-13T12:10:49Z", "mode": "train", "global_step": 1963, "epoch": 0.19718734304369664, "loss": -0.0048, "grad_norm": 10.650731086730957, "learning_rate": 4.054545454545455e-06, "num_tokens": 3528243.0, "completions/mean_length": 38.25, "completions/min_length": 31.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.25, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.9466809034347534, "rewards/meter/std": 0.0385647751390934, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9052823781967163, "rewards/repeat_soft/std": 0.03730171173810959, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.42809373140335083, "rewards/total_composite/std": 0.007216813508421183, "reward": 0.42809373140335083, "reward_std": 0.007216822821646929, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.057822395116090775, "sampling/sampling_logp_difference/max": 0.8509960174560547, "sampling/importance_sampling_ratio/min": 0.42698943614959717, "sampling/importance_sampling_ratio/mean": 1.0015625953674316, "sampling/importance_sampling_ratio/max": 1.9589515924453735, "entropy": 0.2994825579226017, "clip_ratio/low_mean": 0.019611074589192867, "clip_ratio/low_min": 0.019611074589192867, "clip_ratio/high_mean": 0.045088978949934244, "clip_ratio/high_max": 0.045088978949934244, "clip_ratio/region_mean": 0.06470005353912711, "reward_total_mean": 0.42809373140335083, "reward_meter_mean": 0.9466809034347534, "reward_meter_std": 0.0385647751390934, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9052823781967163, "reward_repeat_soft_std": 0.03730171173810959, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.42809373140335083, "reward_total_composite_std": 0.007216813508421183} {"timestamp_utc": "2026-04-13T12:10:56Z", "mode": "train", "global_step": 1964, "epoch": 0.19728779507785033, "loss": -0.0893, "grad_norm": 8.149408340454102, "learning_rate": 4.0515151515151516e-06, "num_tokens": 3530286.0, "completions/mean_length": 91.375, "completions/min_length": 68.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.375, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.8493034243583679, "rewards/meter/std": 0.26503878831863403, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8664325475692749, "rewards/repeat_soft/std": 0.04042574390769005, "rewards/judge_quality/mean": 0.17250001430511475, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.42457038164138794, "rewards/total_composite/std": 0.028597533702850342, "reward": 0.42457038164138794, "reward_std": 0.02859753742814064, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0866965651512146, "sampling/sampling_logp_difference/max": 2.093294620513916, "sampling/importance_sampling_ratio/min": 0.12328030169010162, "sampling/importance_sampling_ratio/mean": 1.0044177770614624, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5302143059670925, "clip_ratio/low_mean": 0.02480624755844474, "clip_ratio/low_min": 0.02480624755844474, "clip_ratio/high_mean": 0.052668992429971695, "clip_ratio/high_max": 0.052668992429971695, "clip_ratio/region_mean": 0.07747523998841643, "reward_total_mean": 0.42457038164138794, "reward_meter_mean": 0.8493034243583679, "reward_meter_std": 0.26503878831863403, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8664325475692749, "reward_repeat_soft_std": 0.04042574390769005, "reward_judge_quality_mean": 0.17250001430511475, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.42457038164138794, "reward_total_composite_std": 0.028597533702850342} {"timestamp_utc": "2026-04-13T12:11:02Z", "mode": "train", "global_step": 1965, "epoch": 0.197388247112004, "loss": -0.0035, "grad_norm": 11.396963119506836, "learning_rate": 4.048484848484849e-06, "num_tokens": 3532160.0, "completions/mean_length": 53.25, "completions/min_length": 42.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.25, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9767234921455383, "rewards/meter/std": 0.008832106366753578, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9145652055740356, "rewards/repeat_soft/std": 0.039254020899534225, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.01603567786514759, "rewards/total_composite/mean": 0.43235844373703003, "rewards/total_composite/std": 0.009502322413027287, "reward": 0.43235844373703003, "reward_std": 0.00950232520699501, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10968401283025742, "sampling/sampling_logp_difference/max": 1.363420009613037, "sampling/importance_sampling_ratio/min": 0.2557845115661621, "sampling/importance_sampling_ratio/mean": 1.0007675886154175, "sampling/importance_sampling_ratio/max": 1.8909069299697876, "entropy": 0.6357946395874023, "clip_ratio/low_mean": 0.04791590222157538, "clip_ratio/low_min": 0.04791590222157538, "clip_ratio/high_mean": 0.05404769070446491, "clip_ratio/high_max": 0.05404769070446491, "clip_ratio/region_mean": 0.10196359292604029, "reward_total_mean": 0.43235844373703003, "reward_meter_mean": 0.9767234921455383, "reward_meter_std": 0.008832106366753578, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9145652055740356, "reward_repeat_soft_std": 0.039254020899534225, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.01603567786514759, "reward_total_composite_mean": 0.43235844373703003, "reward_total_composite_std": 0.009502322413027287} {"timestamp_utc": "2026-04-13T12:11:08Z", "mode": "train", "global_step": 1966, "epoch": 0.19748869914615771, "loss": 0.0788, "grad_norm": 18.476335525512695, "learning_rate": 4.045454545454546e-06, "num_tokens": 3533534.0, "completions/mean_length": 24.75, "completions/min_length": 18.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.75, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.8567436933517456, "rewards/meter/std": 0.16110274195671082, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9362522959709167, "rewards/repeat_soft/std": 0.043951354920864105, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4239703416824341, "rewards/total_composite/std": 0.019093535840511322, "reward": 0.4239703416824341, "reward_std": 0.019093530252575874, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14432452619075775, "sampling/sampling_logp_difference/max": 1.0622243881225586, "sampling/importance_sampling_ratio/min": 0.34568601846694946, "sampling/importance_sampling_ratio/mean": 1.0156381130218506, "sampling/importance_sampling_ratio/max": 1.820920467376709, "entropy": 1.0608180686831474, "clip_ratio/low_mean": 0.05883590504527092, "clip_ratio/low_min": 0.05883590504527092, "clip_ratio/high_mean": 0.10791426338255405, "clip_ratio/high_max": 0.10791426338255405, "clip_ratio/region_mean": 0.16675016842782497, "reward_total_mean": 0.4239703416824341, "reward_meter_mean": 0.8567436933517456, "reward_meter_std": 0.16110274195671082, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9362522959709167, "reward_repeat_soft_std": 0.043951354920864105, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4239703416824341, "reward_total_composite_std": 0.019093535840511322} {"timestamp_utc": "2026-04-13T12:11:15Z", "mode": "train", "global_step": 1967, "epoch": 0.1975891511803114, "loss": -0.0541, "grad_norm": 7.662782192230225, "learning_rate": 4.0424242424242425e-06, "num_tokens": 3535182.0, "completions/mean_length": 50.0, "completions/min_length": 40.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.0, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.906194806098938, "rewards/meter/std": 0.13652828335762024, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.757510244846344, "rewards/repeat_soft/std": 0.11741093546152115, "rewards/judge_quality/mean": 0.14250001311302185, "rewards/judge_quality/std": 0.013887306675314903, "rewards/total_composite/mean": 0.39824041724205017, "rewards/total_composite/std": 0.032019857317209244, "reward": 0.39824041724205017, "reward_std": 0.032019857317209244, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07870042324066162, "sampling/sampling_logp_difference/max": 1.5747723579406738, "sampling/importance_sampling_ratio/min": 0.20705467462539673, "sampling/importance_sampling_ratio/mean": 0.9969842433929443, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.32496722787618637, "clip_ratio/low_mean": 0.00937500037252903, "clip_ratio/low_min": 0.00937500037252903, "clip_ratio/high_mean": 0.06114195752888918, "clip_ratio/high_max": 0.06114195752888918, "clip_ratio/region_mean": 0.07051695790141821, "reward_total_mean": 0.39824041724205017, "reward_meter_mean": 0.906194806098938, "reward_meter_std": 0.13652828335762024, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.757510244846344, "reward_repeat_soft_std": 0.11741093546152115, "reward_judge_quality_mean": 0.14250001311302185, "reward_judge_quality_std": 0.013887306675314903, "reward_total_composite_mean": 0.39824041724205017, "reward_total_composite_std": 0.032019857317209244} {"timestamp_utc": "2026-04-13T12:11:27Z", "mode": "train", "global_step": 1968, "epoch": 0.1976896032144651, "loss": -0.0975, "grad_norm": 2.6602158546447754, "learning_rate": 4.03939393939394e-06, "num_tokens": 3536667.0, "completions/mean_length": 99.625, "completions/min_length": 33.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 40.71428680419922, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.9412317276000977, "rewards/meter/std": 0.09254694730043411, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9318528771400452, "rewards/repeat_soft/std": 0.0765707939863205, "rewards/judge_quality/mean": 0.22875000536441803, "rewards/judge_quality/std": 0.1470119059085846, "rewards/total_composite/mean": 0.43736425042152405, "rewards/total_composite/std": 0.19943781197071075, "reward": 0.43736425042152405, "reward_std": 0.19943781197071075, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12045356631278992, "sampling/sampling_logp_difference/max": 1.456504225730896, "sampling/importance_sampling_ratio/min": 0.23304954171180725, "sampling/importance_sampling_ratio/mean": 0.9907183051109314, "sampling/importance_sampling_ratio/max": 1.699324369430542, "entropy": 0.7500576339662075, "clip_ratio/low_mean": 0.029835973866283894, "clip_ratio/low_min": 0.029835973866283894, "clip_ratio/high_mean": 0.06798899755813181, "clip_ratio/high_max": 0.06798899755813181, "clip_ratio/region_mean": 0.09782497142441571, "reward_total_mean": 0.43736425042152405, "reward_meter_mean": 0.9412317276000977, "reward_meter_std": 0.09254694730043411, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9318528771400452, "reward_repeat_soft_std": 0.0765707939863205, "reward_judge_quality_mean": 0.22875000536441803, "reward_judge_quality_std": 0.1470119059085846, "reward_total_composite_mean": 0.43736425042152405, "reward_total_composite_std": 0.19943781197071075} {"timestamp_utc": "2026-04-13T12:11:34Z", "mode": "train", "global_step": 1969, "epoch": 0.19779005524861878, "loss": -0.0055, "grad_norm": 9.625309944152832, "learning_rate": 4.036363636363637e-06, "num_tokens": 3538634.0, "completions/mean_length": 79.875, "completions/min_length": 67.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.875, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.845735490322113, "rewards/meter/std": 0.24396738409996033, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8271574974060059, "rewards/repeat_soft/std": 0.04683251306414604, "rewards/judge_quality/mean": 0.16124999523162842, "rewards/judge_quality/std": 0.022320717573165894, "rewards/total_composite/mean": 0.4140516519546509, "rewards/total_composite/std": 0.035415295511484146, "reward": 0.4140516519546509, "reward_std": 0.03541528806090355, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10792321711778641, "sampling/sampling_logp_difference/max": 1.4553050994873047, "sampling/importance_sampling_ratio/min": 0.2333291620016098, "sampling/importance_sampling_ratio/mean": 1.0013035535812378, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6829820536077023, "clip_ratio/low_mean": 0.03421615529805422, "clip_ratio/low_min": 0.03421615529805422, "clip_ratio/high_mean": 0.05739721702411771, "clip_ratio/high_max": 0.05739721702411771, "clip_ratio/region_mean": 0.09161337232217193, "reward_total_mean": 0.4140516519546509, "reward_meter_mean": 0.845735490322113, "reward_meter_std": 0.24396738409996033, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8271574974060059, "reward_repeat_soft_std": 0.04683251306414604, "reward_judge_quality_mean": 0.16124999523162842, "reward_judge_quality_std": 0.022320717573165894, "reward_total_composite_mean": 0.4140516519546509, "reward_total_composite_std": 0.035415295511484146} {"timestamp_utc": "2026-04-13T12:11:42Z", "mode": "train", "global_step": 1970, "epoch": 0.19789050728277247, "loss": 0.0563, "grad_norm": 11.966935157775879, "learning_rate": 4.033333333333333e-06, "num_tokens": 3540319.0, "completions/mean_length": 57.625, "completions/min_length": 44.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.625, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.8777778148651123, "rewards/meter/std": 0.2484196424484253, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9263529777526855, "rewards/repeat_soft/std": 0.05342525616288185, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4245362877845764, "rewards/total_composite/std": 0.022866318002343178, "reward": 0.4245362877845764, "reward_std": 0.02286631427705288, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1278618425130844, "sampling/sampling_logp_difference/max": 1.6122455596923828, "sampling/importance_sampling_ratio/min": 0.19943925738334656, "sampling/importance_sampling_ratio/mean": 1.0226764678955078, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9058720171451569, "clip_ratio/low_mean": 0.06446496304124594, "clip_ratio/low_min": 0.06446496304124594, "clip_ratio/high_mean": 0.0880760820582509, "clip_ratio/high_max": 0.0880760820582509, "clip_ratio/region_mean": 0.15254104509949684, "reward_total_mean": 0.4245362877845764, "reward_meter_mean": 0.8777778148651123, "reward_meter_std": 0.2484196424484253, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9263529777526855, "reward_repeat_soft_std": 0.05342525616288185, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4245362877845764, "reward_total_composite_std": 0.022866318002343178} {"timestamp_utc": "2026-04-13T12:11:48Z", "mode": "train", "global_step": 1971, "epoch": 0.19799095931692617, "loss": -0.059, "grad_norm": 9.930120468139648, "learning_rate": 4.030303030303031e-06, "num_tokens": 3542050.0, "completions/mean_length": 56.375, "completions/min_length": 43.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.375, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9330288171768188, "rewards/meter/std": 0.15199299156665802, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9713982939720154, "rewards/repeat_soft/std": 0.0330844447016716, "rewards/judge_quality/mean": 0.1574999988079071, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.4404628872871399, "rewards/total_composite/std": 0.013864600099623203, "reward": 0.4404628872871399, "reward_std": 0.01386459730565548, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10060733556747437, "sampling/sampling_logp_difference/max": 0.9856727123260498, "sampling/importance_sampling_ratio/min": 0.3731880784034729, "sampling/importance_sampling_ratio/mean": 1.014480471611023, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6457661688327789, "clip_ratio/low_mean": 0.04207662586122751, "clip_ratio/low_min": 0.04207662586122751, "clip_ratio/high_mean": 0.0700115580111742, "clip_ratio/high_max": 0.0700115580111742, "clip_ratio/region_mean": 0.11208818387240171, "reward_total_mean": 0.4404628872871399, "reward_meter_mean": 0.9330288171768188, "reward_meter_std": 0.15199299156665802, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9713982939720154, "reward_repeat_soft_std": 0.0330844447016716, "reward_judge_quality_mean": 0.1574999988079071, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.4404628872871399, "reward_total_composite_std": 0.013864600099623203} {"timestamp_utc": "2026-04-13T12:11:55Z", "mode": "train", "global_step": 1972, "epoch": 0.19809141135107985, "loss": -0.0236, "grad_norm": 13.1968412399292, "learning_rate": 4.027272727272727e-06, "num_tokens": 3543993.0, "completions/mean_length": 61.875, "completions/min_length": 56.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.875, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.7542418241500854, "rewards/meter/std": 0.39151984453201294, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9484682083129883, "rewards/repeat_soft/std": 0.04555134102702141, "rewards/judge_quality/mean": 0.1537500023841858, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.41601330041885376, "rewards/total_composite/std": 0.03888684883713722, "reward": 0.41601330041885376, "reward_std": 0.03888684883713722, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11010038107633591, "sampling/sampling_logp_difference/max": 2.005039691925049, "sampling/importance_sampling_ratio/min": 0.1346549540758133, "sampling/importance_sampling_ratio/mean": 1.0076484680175781, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6559870019555092, "clip_ratio/low_mean": 0.01963529083877802, "clip_ratio/low_min": 0.01963529083877802, "clip_ratio/high_mean": 0.07518272940069437, "clip_ratio/high_max": 0.07518272940069437, "clip_ratio/region_mean": 0.09481802023947239, "reward_total_mean": 0.41601330041885376, "reward_meter_mean": 0.7542418241500854, "reward_meter_std": 0.39151984453201294, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9484682083129883, "reward_repeat_soft_std": 0.04555134102702141, "reward_judge_quality_mean": 0.1537500023841858, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.41601330041885376, "reward_total_composite_std": 0.03888684883713722} {"timestamp_utc": "2026-04-13T12:12:01Z", "mode": "train", "global_step": 1973, "epoch": 0.19819186338523356, "loss": -0.0072, "grad_norm": 11.306916236877441, "learning_rate": 4.024242424242424e-06, "num_tokens": 3545615.0, "completions/mean_length": 42.75, "completions/min_length": 38.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.75, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.8588011860847473, "rewards/meter/std": 0.21996483206748962, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.926804780960083, "rewards/repeat_soft/std": 0.06274452805519104, "rewards/judge_quality/mean": 0.3037499785423279, "rewards/judge_quality/std": 0.15665589272975922, "rewards/total_composite/mean": 0.5050768852233887, "rewards/total_composite/std": 0.1065869852900505, "reward": 0.5050768852233887, "reward_std": 0.10658696293830872, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09504842758178711, "sampling/sampling_logp_difference/max": 1.140629768371582, "sampling/importance_sampling_ratio/min": 0.3196176588535309, "sampling/importance_sampling_ratio/mean": 1.0198540687561035, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6223641932010651, "clip_ratio/low_mean": 0.05050806328654289, "clip_ratio/low_min": 0.05050806328654289, "clip_ratio/high_mean": 0.025493421126157045, "clip_ratio/high_max": 0.025493421126157045, "clip_ratio/region_mean": 0.07600148441269994, "reward_total_mean": 0.5050768852233887, "reward_meter_mean": 0.8588011860847473, "reward_meter_std": 0.21996483206748962, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.926804780960083, "reward_repeat_soft_std": 0.06274452805519104, "reward_judge_quality_mean": 0.3037499785423279, "reward_judge_quality_std": 0.15665589272975922, "reward_total_composite_mean": 0.5050768852233887, "reward_total_composite_std": 0.1065869852900505} {"timestamp_utc": "2026-04-13T12:12:07Z", "mode": "train", "global_step": 1974, "epoch": 0.19829231541938724, "loss": 0.0579, "grad_norm": 10.476274490356445, "learning_rate": 4.0212121212121216e-06, "num_tokens": 3547106.0, "completions/mean_length": 39.375, "completions/min_length": 34.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.375, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9804747104644775, "rewards/meter/std": 0.002423355123028159, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9483873844146729, "rewards/repeat_soft/std": 0.027986101806163788, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4378544092178345, "rewards/total_composite/std": 0.004306635819375515, "reward": 0.4378544092178345, "reward_std": 0.0043066395446658134, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12802255153656006, "sampling/sampling_logp_difference/max": 2.76456618309021, "sampling/importance_sampling_ratio/min": 0.06300342828035355, "sampling/importance_sampling_ratio/mean": 0.9863287210464478, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7553790509700775, "clip_ratio/low_mean": 0.02991346619091928, "clip_ratio/low_min": 0.02991346619091928, "clip_ratio/high_mean": 0.08429330121725798, "clip_ratio/high_max": 0.08429330121725798, "clip_ratio/region_mean": 0.11420676740817726, "reward_total_mean": 0.4378544092178345, "reward_meter_mean": 0.9804747104644775, "reward_meter_std": 0.002423355123028159, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9483873844146729, "reward_repeat_soft_std": 0.027986101806163788, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4378544092178345, "reward_total_composite_std": 0.004306635819375515} {"timestamp_utc": "2026-04-13T12:12:14Z", "mode": "train", "global_step": 1975, "epoch": 0.19839276745354092, "loss": 0.0127, "grad_norm": 7.559450626373291, "learning_rate": 4.018181818181818e-06, "num_tokens": 3548907.0, "completions/mean_length": 55.125, "completions/min_length": 45.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.125, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.977043092250824, "rewards/meter/std": 0.012994077056646347, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9380739331245422, "rewards/repeat_soft/std": 0.04333008453249931, "rewards/judge_quality/mean": 0.1875, "rewards/judge_quality/std": 0.1060660108923912, "rewards/total_composite/mean": 0.4601700007915497, "rewards/total_composite/std": 0.07168570160865784, "reward": 0.4601700007915497, "reward_std": 0.07168568670749664, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09898661077022552, "sampling/sampling_logp_difference/max": 1.4066228866577148, "sampling/importance_sampling_ratio/min": 0.2449691742658615, "sampling/importance_sampling_ratio/mean": 1.0076396465301514, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6857492104172707, "clip_ratio/low_mean": 0.08906186185777187, "clip_ratio/low_min": 0.08906186185777187, "clip_ratio/high_mean": 0.010964912362396717, "clip_ratio/high_max": 0.010964912362396717, "clip_ratio/region_mean": 0.10002677422016859, "reward_total_mean": 0.4601700007915497, "reward_meter_mean": 0.977043092250824, "reward_meter_std": 0.012994077056646347, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9380739331245422, "reward_repeat_soft_std": 0.04333008453249931, "reward_judge_quality_mean": 0.1875, "reward_judge_quality_std": 0.1060660108923912, "reward_total_composite_mean": 0.4601700007915497, "reward_total_composite_std": 0.07168570160865784} {"timestamp_utc": "2026-04-13T12:12:21Z", "mode": "train", "global_step": 1976, "epoch": 0.19849321948769463, "loss": -0.0137, "grad_norm": 8.489429473876953, "learning_rate": 4.015151515151515e-06, "num_tokens": 3550904.0, "completions/mean_length": 76.625, "completions/min_length": 67.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.625, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.741714358329773, "rewards/meter/std": 0.2703160345554352, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7749097347259521, "rewards/repeat_soft/std": 0.061277952045202255, "rewards/judge_quality/mean": 0.16500000655651093, "rewards/judge_quality/std": 0.022677872329950333, "rewards/total_composite/mean": 0.39811843633651733, "rewards/total_composite/std": 0.03516698628664017, "reward": 0.39811843633651733, "reward_std": 0.035166990011930466, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0874904990196228, "sampling/sampling_logp_difference/max": 1.5693798065185547, "sampling/importance_sampling_ratio/min": 0.20817424356937408, "sampling/importance_sampling_ratio/mean": 0.9921249747276306, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.39452505111694336, "clip_ratio/low_mean": 0.042162060271948576, "clip_ratio/low_min": 0.042162060271948576, "clip_ratio/high_mean": 0.03482632851228118, "clip_ratio/high_max": 0.03482632851228118, "clip_ratio/region_mean": 0.07698838878422976, "reward_total_mean": 0.39811843633651733, "reward_meter_mean": 0.741714358329773, "reward_meter_std": 0.2703160345554352, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7749097347259521, "reward_repeat_soft_std": 0.061277952045202255, "reward_judge_quality_mean": 0.16500000655651093, "reward_judge_quality_std": 0.022677872329950333, "reward_total_composite_mean": 0.39811843633651733, "reward_total_composite_std": 0.03516698628664017} {"timestamp_utc": "2026-04-13T12:12:29Z", "mode": "train", "global_step": 1977, "epoch": 0.19859367152184831, "loss": 0.0824, "grad_norm": 6.707657814025879, "learning_rate": 4.0121212121212125e-06, "num_tokens": 3553438.0, "completions/mean_length": 138.75, "completions/min_length": 123.0, "completions/max_length": 163.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 138.75, "completions/min_terminated_length": 123.0, "completions/max_terminated_length": 163.0, "rewards/meter/mean": 0.8465820550918579, "rewards/meter/std": 0.2426604926586151, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7107216119766235, "rewards/repeat_soft/std": 0.09084786474704742, "rewards/judge_quality/mean": 0.17250001430511475, "rewards/judge_quality/std": 0.02121320739388466, "rewards/total_composite/mean": 0.40393969416618347, "rewards/total_composite/std": 0.042848143726587296, "reward": 0.40393969416618347, "reward_std": 0.042848147451877594, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06716953217983246, "sampling/sampling_logp_difference/max": 1.9756016731262207, "sampling/importance_sampling_ratio/min": 0.13867785036563873, "sampling/importance_sampling_ratio/mean": 1.003207802772522, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.34012457728385925, "clip_ratio/low_mean": 0.017128449864685535, "clip_ratio/low_min": 0.017128449864685535, "clip_ratio/high_mean": 0.04925884958356619, "clip_ratio/high_max": 0.04925884958356619, "clip_ratio/region_mean": 0.06638729944825172, "reward_total_mean": 0.40393969416618347, "reward_meter_mean": 0.8465820550918579, "reward_meter_std": 0.2426604926586151, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7107216119766235, "reward_repeat_soft_std": 0.09084786474704742, "reward_judge_quality_mean": 0.17250001430511475, "reward_judge_quality_std": 0.02121320739388466, "reward_total_composite_mean": 0.40393969416618347, "reward_total_composite_std": 0.042848143726587296} {"timestamp_utc": "2026-04-13T12:12:36Z", "mode": "train", "global_step": 1978, "epoch": 0.19869412355600202, "loss": 0.0069, "grad_norm": 11.788993835449219, "learning_rate": 4.009090909090909e-06, "num_tokens": 3555132.0, "completions/mean_length": 51.75, "completions/min_length": 46.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.75, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.6127851605415344, "rewards/meter/std": 0.410106360912323, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8013951778411865, "rewards/repeat_soft/std": 0.07675564289093018, "rewards/judge_quality/mean": 0.16500000655651093, "rewards/judge_quality/std": 0.022677872329950333, "rewards/total_composite/mean": 0.3426607847213745, "rewards/total_composite/std": 0.14449571073055267, "reward": 0.3426607847213745, "reward_std": 0.14449572563171387, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11311350017786026, "sampling/sampling_logp_difference/max": 1.820281982421875, "sampling/importance_sampling_ratio/min": 0.1619800627231598, "sampling/importance_sampling_ratio/mean": 1.0193040370941162, "sampling/importance_sampling_ratio/max": 1.9086806774139404, "entropy": 0.7015680745244026, "clip_ratio/low_mean": 0.04053589981049299, "clip_ratio/low_min": 0.04053589981049299, "clip_ratio/high_mean": 0.06337560061365366, "clip_ratio/high_max": 0.06337560061365366, "clip_ratio/region_mean": 0.10391150042414665, "reward_total_mean": 0.3426607847213745, "reward_meter_mean": 0.6127851605415344, "reward_meter_std": 0.410106360912323, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8013951778411865, "reward_repeat_soft_std": 0.07675564289093018, "reward_judge_quality_mean": 0.16500000655651093, "reward_judge_quality_std": 0.022677872329950333, "reward_total_composite_mean": 0.3426607847213745, "reward_total_composite_std": 0.14449571073055267} {"timestamp_utc": "2026-04-13T12:12:42Z", "mode": "train", "global_step": 1979, "epoch": 0.1987945755901557, "loss": 0.005, "grad_norm": 15.920906066894531, "learning_rate": 4.006060606060607e-06, "num_tokens": 3556622.0, "completions/mean_length": 42.25, "completions/min_length": 32.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.25, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.6911098957061768, "rewards/meter/std": 0.3353486657142639, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9723049402236938, "rewards/repeat_soft/std": 0.025577707216143608, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.41322892904281616, "rewards/total_composite/std": 0.032315321266651154, "reward": 0.41322892904281616, "reward_std": 0.03231532871723175, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11365252733230591, "sampling/sampling_logp_difference/max": 1.0233488082885742, "sampling/importance_sampling_ratio/min": 0.35938942432403564, "sampling/importance_sampling_ratio/mean": 1.0163404941558838, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6912884451448917, "clip_ratio/low_mean": 0.04192102886736393, "clip_ratio/low_min": 0.04192102886736393, "clip_ratio/high_mean": 0.06938481144607067, "clip_ratio/high_max": 0.06938481144607067, "clip_ratio/region_mean": 0.1113058403134346, "reward_total_mean": 0.41322892904281616, "reward_meter_mean": 0.6911098957061768, "reward_meter_std": 0.3353486657142639, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9723049402236938, "reward_repeat_soft_std": 0.025577707216143608, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.41322892904281616, "reward_total_composite_std": 0.032315321266651154} {"timestamp_utc": "2026-04-13T12:12:49Z", "mode": "train", "global_step": 1980, "epoch": 0.19889502762430938, "loss": 0.0048, "grad_norm": 12.009119987487793, "learning_rate": 4.003030303030303e-06, "num_tokens": 3558327.0, "completions/mean_length": 58.125, "completions/min_length": 50.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.125, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.7157129049301147, "rewards/meter/std": 0.31780123710632324, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9366830587387085, "rewards/repeat_soft/std": 0.05577899515628815, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.41028445959091187, "rewards/total_composite/std": 0.024794554337859154, "reward": 0.41028445959091187, "reward_std": 0.02479456178843975, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12345890700817108, "sampling/sampling_logp_difference/max": 1.8352665901184082, "sampling/importance_sampling_ratio/min": 0.15957096219062805, "sampling/importance_sampling_ratio/mean": 0.987436830997467, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6226517632603645, "clip_ratio/low_mean": 0.05017365072853863, "clip_ratio/low_min": 0.05017365072853863, "clip_ratio/high_mean": 0.07431062683463097, "clip_ratio/high_max": 0.07431062683463097, "clip_ratio/region_mean": 0.1244842775631696, "reward_total_mean": 0.41028445959091187, "reward_meter_mean": 0.7157129049301147, "reward_meter_std": 0.31780123710632324, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9366830587387085, "reward_repeat_soft_std": 0.05577899515628815, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.41028445959091187, "reward_total_composite_std": 0.024794554337859154} {"timestamp_utc": "2026-04-13T12:12:56Z", "mode": "train", "global_step": 1981, "epoch": 0.1989954796584631, "loss": 0.0349, "grad_norm": 8.917847633361816, "learning_rate": 4.000000000000001e-06, "num_tokens": 3560098.0, "completions/mean_length": 53.375, "completions/min_length": 41.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.375, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.7698092460632324, "rewards/meter/std": 0.38456812500953674, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8634123206138611, "rewards/repeat_soft/std": 0.07371919602155685, "rewards/judge_quality/mean": 0.19500000774860382, "rewards/judge_quality/std": 0.10392305254936218, "rewards/total_composite/mean": 0.40846335887908936, "rewards/total_composite/std": 0.04037882387638092, "reward": 0.40846335887908936, "reward_std": 0.04037882387638092, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1242368221282959, "sampling/sampling_logp_difference/max": 1.6267108917236328, "sampling/importance_sampling_ratio/min": 0.1965750753879547, "sampling/importance_sampling_ratio/mean": 1.0145431756973267, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0416819676756859, "clip_ratio/low_mean": 0.025384615175426006, "clip_ratio/low_min": 0.025384615175426006, "clip_ratio/high_mean": 0.09255419671535492, "clip_ratio/high_max": 0.09255419671535492, "clip_ratio/region_mean": 0.11793881189078093, "reward_total_mean": 0.40846335887908936, "reward_meter_mean": 0.7698092460632324, "reward_meter_std": 0.38456812500953674, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8634123206138611, "reward_repeat_soft_std": 0.07371919602155685, "reward_judge_quality_mean": 0.19500000774860382, "reward_judge_quality_std": 0.10392305254936218, "reward_total_composite_mean": 0.40846335887908936, "reward_total_composite_std": 0.04037882387638092} {"timestamp_utc": "2026-04-13T12:13:03Z", "mode": "train", "global_step": 1982, "epoch": 0.19909593169261677, "loss": 0.0564, "grad_norm": 16.385469436645508, "learning_rate": 3.996969696969698e-06, "num_tokens": 3561620.0, "completions/mean_length": 28.25, "completions/min_length": 26.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.25, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.3600154519081116, "rewards/meter/std": 0.30427849292755127, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.783305287361145, "rewards/repeat_soft/std": 0.05267034471035004, "rewards/judge_quality/mean": 0.1537500023841858, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.35498178005218506, "rewards/total_composite/std": 0.03257571905851364, "reward": 0.35498178005218506, "reward_std": 0.03257571533322334, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11256179958581924, "sampling/sampling_logp_difference/max": 1.1291608810424805, "sampling/importance_sampling_ratio/min": 0.3233044445514679, "sampling/importance_sampling_ratio/mean": 1.009274959564209, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5077512469142675, "clip_ratio/low_mean": 0.051971325650811195, "clip_ratio/low_min": 0.051971325650811195, "clip_ratio/high_mean": 0.07279956713318825, "clip_ratio/high_max": 0.07279956713318825, "clip_ratio/region_mean": 0.12477089278399944, "reward_total_mean": 0.35498178005218506, "reward_meter_mean": 0.3600154519081116, "reward_meter_std": 0.30427849292755127, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.783305287361145, "reward_repeat_soft_std": 0.05267034471035004, "reward_judge_quality_mean": 0.1537500023841858, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.35498178005218506, "reward_total_composite_std": 0.03257571905851364} {"timestamp_utc": "2026-04-13T12:13:15Z", "mode": "train", "global_step": 1983, "epoch": 0.19919638372677045, "loss": -0.2168, "grad_norm": 1.40635347366333, "learning_rate": 3.993939393939394e-06, "num_tokens": 3563997.0, "completions/mean_length": 163.125, "completions/min_length": 91.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 113.28572082519531, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.82413649559021, "rewards/meter/std": 0.3011797070503235, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.21380899846553802, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8315623998641968, "rewards/repeat_soft/std": 0.061598025262355804, "rewards/judge_quality/mean": 0.16375000774860382, "rewards/judge_quality/std": 0.04596194624900818, "rewards/total_composite/mean": 0.37225809693336487, "rewards/total_composite/std": 0.1516929715871811, "reward": 0.37225809693336487, "reward_std": 0.1516929715871811, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07436981052160263, "sampling/sampling_logp_difference/max": 2.8483803272247314, "sampling/importance_sampling_ratio/min": 0.05793808400630951, "sampling/importance_sampling_ratio/mean": 1.0074925422668457, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.39619259908795357, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.06025759922340512, "clip_ratio/high_max": 0.06025759922340512, "clip_ratio/region_mean": 0.06025759922340512, "reward_total_mean": 0.37225809693336487, "reward_meter_mean": 0.82413649559021, "reward_meter_std": 0.3011797070503235, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.21380899846553802, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8315623998641968, "reward_repeat_soft_std": 0.061598025262355804, "reward_judge_quality_mean": 0.16375000774860382, "reward_judge_quality_std": 0.04596194624900818, "reward_total_composite_mean": 0.37225809693336487, "reward_total_composite_std": 0.1516929715871811} {"timestamp_utc": "2026-04-13T12:13:22Z", "mode": "train", "global_step": 1984, "epoch": 0.19929683576092416, "loss": -0.0194, "grad_norm": 10.250814437866211, "learning_rate": 3.990909090909092e-06, "num_tokens": 3565772.0, "completions/mean_length": 46.875, "completions/min_length": 40.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.875, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.8747069835662842, "rewards/meter/std": 0.2825613021850586, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.975144624710083, "rewards/repeat_soft/std": 0.02640417404472828, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4315556287765503, "rewards/total_composite/std": 0.027250397950410843, "reward": 0.4315556287765503, "reward_std": 0.02725040167570114, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13391144573688507, "sampling/sampling_logp_difference/max": 2.1814565658569336, "sampling/importance_sampling_ratio/min": 0.1128769963979721, "sampling/importance_sampling_ratio/mean": 1.0205755233764648, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8451790362596512, "clip_ratio/low_mean": 0.022619048599153757, "clip_ratio/low_min": 0.022619048599153757, "clip_ratio/high_mean": 0.0787371383048594, "clip_ratio/high_max": 0.0787371383048594, "clip_ratio/region_mean": 0.10135618690401316, "reward_total_mean": 0.4315556287765503, "reward_meter_mean": 0.8747069835662842, "reward_meter_std": 0.2825613021850586, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.975144624710083, "reward_repeat_soft_std": 0.02640417404472828, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4315556287765503, "reward_total_composite_std": 0.027250397950410843} {"timestamp_utc": "2026-04-13T12:13:29Z", "mode": "train", "global_step": 1985, "epoch": 0.19939728779507784, "loss": 0.0396, "grad_norm": 8.761999130249023, "learning_rate": 3.987878787878788e-06, "num_tokens": 3567525.0, "completions/mean_length": 57.125, "completions/min_length": 36.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.125, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.8532387018203735, "rewards/meter/std": 0.30601271986961365, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9405505657196045, "rewards/repeat_soft/std": 0.04098353907465935, "rewards/judge_quality/mean": 0.26125001907348633, "rewards/judge_quality/std": 0.2665889263153076, "rewards/total_composite/mean": 0.4926952123641968, "rewards/total_composite/std": 0.1742052137851715, "reward": 0.4926952123641968, "reward_std": 0.1742052137851715, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10535774379968643, "sampling/sampling_logp_difference/max": 1.3276968002319336, "sampling/importance_sampling_ratio/min": 0.2650870978832245, "sampling/importance_sampling_ratio/mean": 0.9956843256950378, "sampling/importance_sampling_ratio/max": 1.956322431564331, "entropy": 0.6120221167802811, "clip_ratio/low_mean": 0.046632946934551, "clip_ratio/low_min": 0.046632946934551, "clip_ratio/high_mean": 0.014423076994717121, "clip_ratio/high_max": 0.014423076994717121, "clip_ratio/region_mean": 0.06105602392926812, "reward_total_mean": 0.4926952123641968, "reward_meter_mean": 0.8532387018203735, "reward_meter_std": 0.30601271986961365, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9405505657196045, "reward_repeat_soft_std": 0.04098353907465935, "reward_judge_quality_mean": 0.26125001907348633, "reward_judge_quality_std": 0.2665889263153076, "reward_total_composite_mean": 0.4926952123641968, "reward_total_composite_std": 0.1742052137851715} {"timestamp_utc": "2026-04-13T12:13:35Z", "mode": "train", "global_step": 1986, "epoch": 0.19949773982923155, "loss": 0.0726, "grad_norm": 7.136348724365234, "learning_rate": 3.984848484848485e-06, "num_tokens": 3569292.0, "completions/mean_length": 46.875, "completions/min_length": 39.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.875, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9778975248336792, "rewards/meter/std": 0.011393287219107151, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9715107679367065, "rewards/repeat_soft/std": 0.03862692788243294, "rewards/judge_quality/mean": 0.20250001549720764, "rewards/judge_quality/std": 0.10110107809305191, "rewards/total_composite/mean": 0.47444748878479004, "rewards/total_composite/std": 0.06477969139814377, "reward": 0.47444748878479004, "reward_std": 0.06477969139814377, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12436336278915405, "sampling/sampling_logp_difference/max": 2.2677464485168457, "sampling/importance_sampling_ratio/min": 0.10354526340961456, "sampling/importance_sampling_ratio/mean": 1.0028719902038574, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5972497388720512, "clip_ratio/low_mean": 0.11246980540454388, "clip_ratio/low_min": 0.11246980540454388, "clip_ratio/high_mean": 0.01923076994717121, "clip_ratio/high_max": 0.01923076994717121, "clip_ratio/region_mean": 0.1317005753517151, "reward_total_mean": 0.47444748878479004, "reward_meter_mean": 0.9778975248336792, "reward_meter_std": 0.011393287219107151, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9715107679367065, "reward_repeat_soft_std": 0.03862692788243294, "reward_judge_quality_mean": 0.20250001549720764, "reward_judge_quality_std": 0.10110107809305191, "reward_total_composite_mean": 0.47444748878479004, "reward_total_composite_std": 0.06477969139814377} {"timestamp_utc": "2026-04-13T12:13:42Z", "mode": "train", "global_step": 1987, "epoch": 0.19959819186338523, "loss": 0.0538, "grad_norm": 11.674086570739746, "learning_rate": 3.9818181818181825e-06, "num_tokens": 3571052.0, "completions/mean_length": 57.0, "completions/min_length": 48.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.0, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.8795936107635498, "rewards/meter/std": 0.15953429043293, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9084420204162598, "rewards/repeat_soft/std": 0.044956889003515244, "rewards/judge_quality/mean": 0.1574999988079071, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.42602890729904175, "rewards/total_composite/std": 0.015605825930833817, "reward": 0.42602890729904175, "reward_std": 0.015605824068188667, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09734667837619781, "sampling/sampling_logp_difference/max": 2.8360369205474854, "sampling/importance_sampling_ratio/min": 0.05865767225623131, "sampling/importance_sampling_ratio/mean": 1.0042951107025146, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5530225820839405, "clip_ratio/low_mean": 0.06306716450490057, "clip_ratio/low_min": 0.06306716450490057, "clip_ratio/high_mean": 0.04177061468362808, "clip_ratio/high_max": 0.04177061468362808, "clip_ratio/region_mean": 0.10483777918852866, "reward_total_mean": 0.42602890729904175, "reward_meter_mean": 0.8795936107635498, "reward_meter_std": 0.15953429043293, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9084420204162598, "reward_repeat_soft_std": 0.044956889003515244, "reward_judge_quality_mean": 0.1574999988079071, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.42602890729904175, "reward_total_composite_std": 0.015605825930833817} {"timestamp_utc": "2026-04-13T12:13:49Z", "mode": "train", "global_step": 1988, "epoch": 0.1996986438975389, "loss": -0.0229, "grad_norm": 5.305408477783203, "learning_rate": 3.978787878787879e-06, "num_tokens": 3573749.0, "completions/mean_length": 139.125, "completions/min_length": 118.0, "completions/max_length": 155.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 139.125, "completions/min_terminated_length": 118.0, "completions/max_terminated_length": 155.0, "rewards/meter/mean": 0.9411613941192627, "rewards/meter/std": 0.06411188840866089, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7919429540634155, "rewards/repeat_soft/std": 0.04169183969497681, "rewards/judge_quality/mean": 0.17250001430511475, "rewards/judge_quality/std": 0.02121320739388466, "rewards/total_composite/mean": 0.4193919599056244, "rewards/total_composite/std": 0.02630588412284851, "reward": 0.4193919599056244, "reward_std": 0.02630588412284851, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07875581830739975, "sampling/sampling_logp_difference/max": 2.4715664386749268, "sampling/importance_sampling_ratio/min": 0.08445246517658234, "sampling/importance_sampling_ratio/mean": 1.006574273109436, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.48092228919267654, "clip_ratio/low_mean": 0.016903753392398357, "clip_ratio/low_min": 0.016903753392398357, "clip_ratio/high_mean": 0.06237533036619425, "clip_ratio/high_max": 0.06237533036619425, "clip_ratio/region_mean": 0.0792790837585926, "reward_total_mean": 0.4193919599056244, "reward_meter_mean": 0.9411613941192627, "reward_meter_std": 0.06411188840866089, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7919429540634155, "reward_repeat_soft_std": 0.04169183969497681, "reward_judge_quality_mean": 0.17250001430511475, "reward_judge_quality_std": 0.02121320739388466, "reward_total_composite_mean": 0.4193919599056244, "reward_total_composite_std": 0.02630588412284851} {"timestamp_utc": "2026-04-13T12:13:56Z", "mode": "train", "global_step": 1989, "epoch": 0.19979909593169262, "loss": -0.0049, "grad_norm": 6.995320796966553, "learning_rate": 3.975757575757576e-06, "num_tokens": 3575843.0, "completions/mean_length": 84.75, "completions/min_length": 69.0, "completions/max_length": 111.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.75, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 111.0, "rewards/meter/mean": 0.899039626121521, "rewards/meter/std": 0.14101403951644897, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.815096378326416, "rewards/repeat_soft/std": 0.057892054319381714, "rewards/judge_quality/mean": 0.1612500101327896, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.4156867563724518, "rewards/total_composite/std": 0.017065096646547318, "reward": 0.4156867563724518, "reward_std": 0.017065098509192467, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11169660091400146, "sampling/sampling_logp_difference/max": 1.1490683555603027, "sampling/importance_sampling_ratio/min": 0.31693190336227417, "sampling/importance_sampling_ratio/mean": 1.0052859783172607, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.726189449429512, "clip_ratio/low_mean": 0.03251077188178897, "clip_ratio/low_min": 0.03251077188178897, "clip_ratio/high_mean": 0.07103743311017752, "clip_ratio/high_max": 0.07103743311017752, "clip_ratio/region_mean": 0.10354820499196649, "reward_total_mean": 0.4156867563724518, "reward_meter_mean": 0.899039626121521, "reward_meter_std": 0.14101403951644897, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.815096378326416, "reward_repeat_soft_std": 0.057892054319381714, "reward_judge_quality_mean": 0.1612500101327896, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.4156867563724518, "reward_total_composite_std": 0.017065096646547318} {"timestamp_utc": "2026-04-13T12:14:08Z", "mode": "train", "global_step": 1990, "epoch": 0.1998995479658463, "loss": -0.21, "grad_norm": 1.948974370956421, "learning_rate": 3.972727272727273e-06, "num_tokens": 3577916.0, "completions/mean_length": 209.125, "completions/min_length": 87.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 108.16667175292969, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.6804710626602173, "rewards/meter/std": 0.34284526109695435, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9123061895370483, "rewards/repeat_soft/std": 0.0672207847237587, "rewards/judge_quality/mean": 0.1287499964237213, "rewards/judge_quality/std": 0.05462535098195076, "rewards/total_composite/mean": 0.30792689323425293, "rewards/total_composite/std": 0.19347305595874786, "reward": 0.30792689323425293, "reward_std": 0.19347302615642548, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11115031689405441, "sampling/sampling_logp_difference/max": 1.3480873107910156, "sampling/importance_sampling_ratio/min": 0.259736567735672, "sampling/importance_sampling_ratio/mean": 1.0218712091445923, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6053452044725418, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.07028499897569418, "clip_ratio/high_max": 0.07028499897569418, "clip_ratio/region_mean": 0.07028499897569418, "reward_total_mean": 0.30792689323425293, "reward_meter_mean": 0.6804710626602173, "reward_meter_std": 0.34284526109695435, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9123061895370483, "reward_repeat_soft_std": 0.0672207847237587, "reward_judge_quality_mean": 0.1287499964237213, "reward_judge_quality_std": 0.05462535098195076, "reward_total_composite_mean": 0.30792689323425293, "reward_total_composite_std": 0.19347305595874786} {"timestamp_utc": "2026-04-13T12:14:15Z", "mode": "train", "global_step": 1991, "epoch": 0.2, "loss": -0.0089, "grad_norm": 10.88731575012207, "learning_rate": 3.96969696969697e-06, "num_tokens": 3579557.0, "completions/mean_length": 45.125, "completions/min_length": 40.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.125, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.9664255380630493, "rewards/meter/std": 0.015298202633857727, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.955539345741272, "rewards/repeat_soft/std": 0.03646053746342659, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4375573992729187, "rewards/total_composite/std": 0.005865698214620352, "reward": 0.4375573992729187, "reward_std": 0.005865701008588076, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16127146780490875, "sampling/sampling_logp_difference/max": 1.6951549053192139, "sampling/importance_sampling_ratio/min": 0.18357078731060028, "sampling/importance_sampling_ratio/mean": 1.020660400390625, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9290966093540192, "clip_ratio/low_mean": 0.06766532361507416, "clip_ratio/low_min": 0.06766532361507416, "clip_ratio/high_mean": 0.08113236539065838, "clip_ratio/high_max": 0.08113236539065838, "clip_ratio/region_mean": 0.14879768900573254, "reward_total_mean": 0.4375573992729187, "reward_meter_mean": 0.9664255380630493, "reward_meter_std": 0.015298202633857727, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.955539345741272, "reward_repeat_soft_std": 0.03646053746342659, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4375573992729187, "reward_total_composite_std": 0.005865698214620352} {"timestamp_utc": "2026-04-13T12:14:22Z", "mode": "train", "global_step": 1992, "epoch": 0.2001004520341537, "loss": 0.0962, "grad_norm": 8.176782608032227, "learning_rate": 3.966666666666667e-06, "num_tokens": 3581503.0, "completions/mean_length": 80.25, "completions/min_length": 66.0, "completions/max_length": 100.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.25, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 100.0, "rewards/meter/mean": 0.9668633937835693, "rewards/meter/std": 0.058630939573049545, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7522540092468262, "rewards/repeat_soft/std": 0.046631213277578354, "rewards/judge_quality/mean": 0.17250001430511475, "rewards/judge_quality/std": 0.02121320739388466, "rewards/total_composite/mean": 0.4156605899333954, "rewards/total_composite/std": 0.037274960428476334, "reward": 0.4156605899333954, "reward_std": 0.037274960428476334, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08009006828069687, "sampling/sampling_logp_difference/max": 1.8451592922210693, "sampling/importance_sampling_ratio/min": 0.1580001562833786, "sampling/importance_sampling_ratio/mean": 1.0078073740005493, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5122839771211147, "clip_ratio/low_mean": 0.004999999888241291, "clip_ratio/low_min": 0.004999999888241291, "clip_ratio/high_mean": 0.06474924786016345, "clip_ratio/high_max": 0.06474924786016345, "clip_ratio/region_mean": 0.06974924774840474, "reward_total_mean": 0.4156605899333954, "reward_meter_mean": 0.9668633937835693, "reward_meter_std": 0.058630939573049545, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7522540092468262, "reward_repeat_soft_std": 0.046631213277578354, "reward_judge_quality_mean": 0.17250001430511475, "reward_judge_quality_std": 0.02121320739388466, "reward_total_composite_mean": 0.4156605899333954, "reward_total_composite_std": 0.037274960428476334} {"timestamp_utc": "2026-04-13T12:14:29Z", "mode": "train", "global_step": 1993, "epoch": 0.20020090406830737, "loss": 0.0327, "grad_norm": 11.174549102783203, "learning_rate": 3.963636363636364e-06, "num_tokens": 3583373.0, "completions/mean_length": 66.75, "completions/min_length": 55.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.75, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.8860222697257996, "rewards/meter/std": 0.2954499423503876, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9293386936187744, "rewards/repeat_soft/std": 0.05199816823005676, "rewards/judge_quality/mean": 0.17250001430511475, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.438253790140152, "rewards/total_composite/std": 0.03380537033081055, "reward": 0.438253790140152, "reward_std": 0.03380536660552025, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11163672804832458, "sampling/sampling_logp_difference/max": 2.6584410667419434, "sampling/importance_sampling_ratio/min": 0.07005735486745834, "sampling/importance_sampling_ratio/mean": 0.9926469922065735, "sampling/importance_sampling_ratio/max": 1.786298155784607, "entropy": 0.5487346686422825, "clip_ratio/low_mean": 0.022989511489868164, "clip_ratio/low_min": 0.022989511489868164, "clip_ratio/high_mean": 0.08296406362205744, "clip_ratio/high_max": 0.08296406362205744, "clip_ratio/region_mean": 0.1059535751119256, "reward_total_mean": 0.438253790140152, "reward_meter_mean": 0.8860222697257996, "reward_meter_std": 0.2954499423503876, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9293386936187744, "reward_repeat_soft_std": 0.05199816823005676, "reward_judge_quality_mean": 0.17250001430511475, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.438253790140152, "reward_total_composite_std": 0.03380537033081055} {"timestamp_utc": "2026-04-13T12:14:35Z", "mode": "train", "global_step": 1994, "epoch": 0.20030135610246108, "loss": -0.1128, "grad_norm": 19.808204650878906, "learning_rate": 3.960606060606061e-06, "num_tokens": 3584738.0, "completions/mean_length": 18.625, "completions/min_length": 15.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 18.625, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.17641128599643707, "rewards/meter/std": 0.30511367321014404, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9604669809341431, "rewards/repeat_soft/std": 0.005061782896518707, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.36127012968063354, "rewards/total_composite/std": 0.029887478798627853, "reward": 0.36127012968063354, "reward_std": 0.029887471348047256, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13217005133628845, "sampling/sampling_logp_difference/max": 1.6605939865112305, "sampling/importance_sampling_ratio/min": 0.19002607464790344, "sampling/importance_sampling_ratio/mean": 1.0037096738815308, "sampling/importance_sampling_ratio/max": 1.9682484865188599, "entropy": 0.8014398291707039, "clip_ratio/low_mean": 0.10888788476586342, "clip_ratio/low_min": 0.10888788476586342, "clip_ratio/high_mean": 0.01923076994717121, "clip_ratio/high_max": 0.01923076994717121, "clip_ratio/region_mean": 0.12811865471303463, "reward_total_mean": 0.36127012968063354, "reward_meter_mean": 0.17641128599643707, "reward_meter_std": 0.30511367321014404, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9604669809341431, "reward_repeat_soft_std": 0.005061782896518707, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.36127012968063354, "reward_total_composite_std": 0.029887478798627853} {"timestamp_utc": "2026-04-13T12:14:42Z", "mode": "train", "global_step": 1995, "epoch": 0.20040180813661476, "loss": -0.0115, "grad_norm": 10.272811889648438, "learning_rate": 3.957575757575758e-06, "num_tokens": 3586344.0, "completions/mean_length": 39.75, "completions/min_length": 37.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.75, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.9224235415458679, "rewards/meter/std": 0.11386893689632416, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9877801537513733, "rewards/repeat_soft/std": 0.024328239262104034, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4381033182144165, "rewards/total_composite/std": 0.010988668538630009, "reward": 0.4381033182144165, "reward_std": 0.010988662950694561, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08340685069561005, "sampling/sampling_logp_difference/max": 2.556256055831909, "sampling/importance_sampling_ratio/min": 0.07759470492601395, "sampling/importance_sampling_ratio/mean": 1.008427381515503, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.31179263442754745, "clip_ratio/low_mean": 0.016372141893953085, "clip_ratio/low_min": 0.016372141893953085, "clip_ratio/high_mean": 0.0549057312309742, "clip_ratio/high_max": 0.0549057312309742, "clip_ratio/region_mean": 0.07127787312492728, "reward_total_mean": 0.4381033182144165, "reward_meter_mean": 0.9224235415458679, "reward_meter_std": 0.11386893689632416, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9877801537513733, "reward_repeat_soft_std": 0.024328239262104034, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4381033182144165, "reward_total_composite_std": 0.010988668538630009} {"timestamp_utc": "2026-04-13T12:14:53Z", "mode": "train", "global_step": 1996, "epoch": 0.20050226017076847, "loss": -0.1921, "grad_norm": 1.6462302207946777, "learning_rate": 3.954545454545454e-06, "num_tokens": 3588464.0, "completions/mean_length": 149.0, "completions/min_length": 77.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 97.14286041259766, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 127.0, "rewards/meter/mean": 0.9769530296325684, "rewards/meter/std": 0.0088235167786479, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8973762392997742, "rewards/repeat_soft/std": 0.033636774867773056, "rewards/judge_quality/mean": 0.13750000298023224, "rewards/judge_quality/std": 0.04200340807437897, "rewards/total_composite/mean": 0.3751123547554016, "rewards/total_composite/std": 0.15231961011886597, "reward": 0.3751123547554016, "reward_std": 0.15231962502002716, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09448987245559692, "sampling/sampling_logp_difference/max": 1.3202617168426514, "sampling/importance_sampling_ratio/min": 0.2670654058456421, "sampling/importance_sampling_ratio/mean": 1.007238745689392, "sampling/importance_sampling_ratio/max": 1.8536213636398315, "entropy": 0.5517528206110001, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.07715445989742875, "clip_ratio/high_max": 0.07715445989742875, "clip_ratio/region_mean": 0.07715445989742875, "reward_total_mean": 0.3751123547554016, "reward_meter_mean": 0.9769530296325684, "reward_meter_std": 0.0088235167786479, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8973762392997742, "reward_repeat_soft_std": 0.033636774867773056, "reward_judge_quality_mean": 0.13750000298023224, "reward_judge_quality_std": 0.04200340807437897, "reward_total_composite_mean": 0.3751123547554016, "reward_total_composite_std": 0.15231961011886597} {"timestamp_utc": "2026-04-13T12:15:00Z", "mode": "train", "global_step": 1997, "epoch": 0.20060271220492215, "loss": 0.0003, "grad_norm": 10.457941055297852, "learning_rate": 3.951515151515152e-06, "num_tokens": 3590367.0, "completions/mean_length": 51.875, "completions/min_length": 47.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.875, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.8365335464477539, "rewards/meter/std": 0.21539472043514252, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8633731603622437, "rewards/repeat_soft/std": 0.12307754158973694, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.01603567786514759, "rewards/total_composite/mean": 0.36861011385917664, "rewards/total_composite/std": 0.15087535977363586, "reward": 0.36861011385917664, "reward_std": 0.15087535977363586, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10408765077590942, "sampling/sampling_logp_difference/max": 1.6148912906646729, "sampling/importance_sampling_ratio/min": 0.19891229271888733, "sampling/importance_sampling_ratio/mean": 1.0068790912628174, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4990379549562931, "clip_ratio/low_mean": 0.0078125, "clip_ratio/low_min": 0.0078125, "clip_ratio/high_mean": 0.07741094008088112, "clip_ratio/high_max": 0.07741094008088112, "clip_ratio/region_mean": 0.08522344008088112, "reward_total_mean": 0.36861011385917664, "reward_meter_mean": 0.8365335464477539, "reward_meter_std": 0.21539472043514252, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8633731603622437, "reward_repeat_soft_std": 0.12307754158973694, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.01603567786514759, "reward_total_composite_mean": 0.36861011385917664, "reward_total_composite_std": 0.15087535977363586} {"timestamp_utc": "2026-04-13T12:15:06Z", "mode": "train", "global_step": 1998, "epoch": 0.20070316423907583, "loss": 0.0428, "grad_norm": 5.65458869934082, "learning_rate": 3.948484848484849e-06, "num_tokens": 3591879.0, "completions/mean_length": 69.0, "completions/min_length": 58.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.0, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.9892727136611938, "rewards/meter/std": 0.008448584005236626, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7639752626419067, "rewards/repeat_soft/std": 0.051463864743709564, "rewards/judge_quality/mean": 0.1574999988079071, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.4158993363380432, "rewards/total_composite/std": 0.01525058876723051, "reward": 0.4158993363380432, "reward_std": 0.015250589698553085, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07780607044696808, "sampling/sampling_logp_difference/max": 1.2431490421295166, "sampling/importance_sampling_ratio/min": 0.28847435116767883, "sampling/importance_sampling_ratio/mean": 1.0062282085418701, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.650130495429039, "clip_ratio/low_mean": 0.02444823505356908, "clip_ratio/low_min": 0.02444823505356908, "clip_ratio/high_mean": 0.04893648810684681, "clip_ratio/high_max": 0.04893648810684681, "clip_ratio/region_mean": 0.07338472316041589, "reward_total_mean": 0.4158993363380432, "reward_meter_mean": 0.9892727136611938, "reward_meter_std": 0.008448584005236626, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7639752626419067, "reward_repeat_soft_std": 0.051463864743709564, "reward_judge_quality_mean": 0.1574999988079071, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.4158993363380432, "reward_total_composite_std": 0.01525058876723051} {"timestamp_utc": "2026-04-13T12:15:13Z", "mode": "train", "global_step": 1999, "epoch": 0.20080361627322954, "loss": 0.0096, "grad_norm": 5.741130352020264, "learning_rate": 3.945454545454545e-06, "num_tokens": 3594125.0, "completions/mean_length": 103.75, "completions/min_length": 87.0, "completions/max_length": 123.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 103.75, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.9535378813743591, "rewards/meter/std": 0.04614168405532837, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7956213355064392, "rewards/repeat_soft/std": 0.03852888196706772, "rewards/judge_quality/mean": 0.17250001430511475, "rewards/judge_quality/std": 0.02121320739388466, "rewards/total_composite/mean": 0.41989007592201233, "rewards/total_composite/std": 0.022876359522342682, "reward": 0.41989007592201233, "reward_std": 0.02287636697292328, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08122720569372177, "sampling/sampling_logp_difference/max": 1.6818737983703613, "sampling/importance_sampling_ratio/min": 0.18602506816387177, "sampling/importance_sampling_ratio/mean": 1.006203055381775, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4709929823875427, "clip_ratio/low_mean": 0.02482958883047104, "clip_ratio/low_min": 0.02482958883047104, "clip_ratio/high_mean": 0.06267943885177374, "clip_ratio/high_max": 0.06267943885177374, "clip_ratio/region_mean": 0.08750902768224478, "reward_total_mean": 0.41989007592201233, "reward_meter_mean": 0.9535378813743591, "reward_meter_std": 0.04614168405532837, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7956213355064392, "reward_repeat_soft_std": 0.03852888196706772, "reward_judge_quality_mean": 0.17250001430511475, "reward_judge_quality_std": 0.02121320739388466, "reward_total_composite_mean": 0.41989007592201233, "reward_total_composite_std": 0.022876359522342682} {"timestamp_utc": "2026-04-13T12:15:20Z", "mode": "train", "global_step": 2000, "epoch": 0.20090406830738322, "loss": 0.0798, "grad_norm": 16.039695739746094, "learning_rate": 3.942424242424243e-06, "num_tokens": 3595641.0, "completions/mean_length": 28.5, "completions/min_length": 23.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.5, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9656250476837158, "rewards/meter/std": 0.038042064756155014, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.921393871307373, "rewards/repeat_soft/std": 0.04221487417817116, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.432357519865036, "rewards/total_composite/std": 0.006993249524384737, "reward": 0.432357519865036, "reward_std": 0.0069932518526911736, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09250590950250626, "sampling/sampling_logp_difference/max": 1.1924142837524414, "sampling/importance_sampling_ratio/min": 0.3034876883029938, "sampling/importance_sampling_ratio/mean": 1.0109021663665771, "sampling/importance_sampling_ratio/max": 1.5811668634414673, "entropy": 0.5452602319419384, "clip_ratio/low_mean": 0.03977534594014287, "clip_ratio/low_min": 0.03977534594014287, "clip_ratio/high_mean": 0.04168054927140474, "clip_ratio/high_max": 0.04168054927140474, "clip_ratio/region_mean": 0.08145589521154761, "reward_total_mean": 0.432357519865036, "reward_meter_mean": 0.9656250476837158, "reward_meter_std": 0.038042064756155014, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.921393871307373, "reward_repeat_soft_std": 0.04221487417817116, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.432357519865036, "reward_total_composite_std": 0.006993249524384737} {"timestamp_utc": "2026-04-13T12:16:12Z", "mode": "eval", "global_step": 2000, "epoch": 0.20090406830738322, "eval_loss": NaN, "eval_runtime": 51.9949, "eval_samples_per_second": 1.539, "eval_steps_per_second": 0.192, "eval_num_tokens": 3595641.0, "eval_completions/mean_length": 87.6125, "eval_completions/min_length": 34.1, "eval_completions/max_length": 222.1, "eval_completions/clipped_ratio": 0.025, "eval_completions/mean_terminated_length": 76.95892944335938, "eval_completions/min_terminated_length": 34.1, "eval_completions/max_terminated_length": 144.4, "eval_rewards/meter/mean": 0.8989010989665985, "eval_rewards/meter/std": 0.193228018283844, "eval_rewards/count_adherence/mean": 0.9708333313465118, "eval_rewards/count_adherence/std": 0.045636449754238126, "eval_rewards/hard_gate/mean": 0.975, "eval_rewards/hard_gate/std": 0.07071067690849304, "eval_rewards/repeat_soft/mean": 0.8757998764514923, "eval_rewards/repeat_soft/std": 0.08093975968658924, "eval_rewards/judge_quality/mean": 0.16025000661611558, "eval_rewards/judge_quality/std": 0.03135918769985437, "eval_rewards/total_composite/mean": 0.41270904541015624, "eval_rewards/total_composite/std": 0.05069283861666918, "eval_reward": 0.41270904541015624, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.051789015904068945, "eval_sampling/sampling_logp_difference/max": 1.1813788414001465, "eval_sampling/importance_sampling_ratio/min": 0.31964881867170336, "eval_sampling/importance_sampling_ratio/mean": 1.0109478116035462, "eval_sampling/importance_sampling_ratio/max": 1.473687183856964, "eval_entropy": 0.5352312296628952, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.41270904541015624, "eval_reward_meter_mean": 0.8989010989665985, "eval_reward_meter_std": 0.193228018283844, "eval_reward_count_adherence_mean": 0.9708333313465118, "eval_reward_count_adherence_std": 0.045636449754238126, "eval_reward_hard_gate_mean": 0.975, "eval_reward_hard_gate_std": 0.07071067690849304, "eval_reward_repeat_soft_mean": 0.8757998764514923, "eval_reward_repeat_soft_std": 0.08093975968658924, "eval_reward_judge_quality_mean": 0.16025000661611558, "eval_reward_judge_quality_std": 0.03135918769985437, "eval_reward_total_composite_mean": 0.41270904541015624, "eval_reward_total_composite_std": 0.05069283861666918} {"timestamp_utc": "2026-04-13T12:16:21Z", "mode": "train", "global_step": 2001, "epoch": 0.20100452034153693, "loss": 0.0855, "grad_norm": 7.101884841918945, "learning_rate": 3.93939393939394e-06, "num_tokens": 3597319.0, "completions/mean_length": 51.75, "completions/min_length": 42.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.75, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.942493200302124, "rewards/meter/std": 0.07095000147819519, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9551660418510437, "rewards/repeat_soft/std": 0.013545302674174309, "rewards/judge_quality/mean": 0.1875, "rewards/judge_quality/std": 0.1060660108923912, "rewards/total_composite/mean": 0.458089143037796, "rewards/total_composite/std": 0.06420551985502243, "reward": 0.458089143037796, "reward_std": 0.06420551240444183, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08462902158498764, "sampling/sampling_logp_difference/max": 1.5644030570983887, "sampling/importance_sampling_ratio/min": 0.20921286940574646, "sampling/importance_sampling_ratio/mean": 1.0072275400161743, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4804040603339672, "clip_ratio/low_mean": 0.054138449020683765, "clip_ratio/low_min": 0.054138449020683765, "clip_ratio/high_mean": 0.011904762126505375, "clip_ratio/high_max": 0.011904762126505375, "clip_ratio/region_mean": 0.06604321114718914, "reward_total_mean": 0.458089143037796, "reward_meter_mean": 0.942493200302124, "reward_meter_std": 0.07095000147819519, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9551660418510437, "reward_repeat_soft_std": 0.013545302674174309, "reward_judge_quality_mean": 0.1875, "reward_judge_quality_std": 0.1060660108923912, "reward_total_composite_mean": 0.458089143037796, "reward_total_composite_std": 0.06420551985502243} {"timestamp_utc": "2026-04-13T12:16:29Z", "mode": "train", "global_step": 2002, "epoch": 0.2011049723756906, "loss": 0.0188, "grad_norm": 8.576906204223633, "learning_rate": 3.936363636363636e-06, "num_tokens": 3599182.0, "completions/mean_length": 70.875, "completions/min_length": 65.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.875, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.8849501609802246, "rewards/meter/std": 0.2143920212984085, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9152114391326904, "rewards/repeat_soft/std": 0.05425975099205971, "rewards/judge_quality/mean": 0.16124999523162842, "rewards/judge_quality/std": 0.022320717573165894, "rewards/total_composite/mean": 0.42890387773513794, "rewards/total_composite/std": 0.02358926832675934, "reward": 0.42890387773513794, "reward_std": 0.02358926832675934, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09013252705335617, "sampling/sampling_logp_difference/max": 1.0857977867126465, "sampling/importance_sampling_ratio/min": 0.33763232827186584, "sampling/importance_sampling_ratio/mean": 1.0102643966674805, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4903973340988159, "clip_ratio/low_mean": 0.057537827640771866, "clip_ratio/low_min": 0.057537827640771866, "clip_ratio/high_mean": 0.05734657868742943, "clip_ratio/high_max": 0.05734657868742943, "clip_ratio/region_mean": 0.1148844063282013, "reward_total_mean": 0.42890387773513794, "reward_meter_mean": 0.8849501609802246, "reward_meter_std": 0.2143920212984085, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9152114391326904, "reward_repeat_soft_std": 0.05425975099205971, "reward_judge_quality_mean": 0.16124999523162842, "reward_judge_quality_std": 0.022320717573165894, "reward_total_composite_mean": 0.42890387773513794, "reward_total_composite_std": 0.02358926832675934} {"timestamp_utc": "2026-04-13T12:16:41Z", "mode": "train", "global_step": 2003, "epoch": 0.2012054244098443, "loss": -0.1107, "grad_norm": 2.2421905994415283, "learning_rate": 3.9333333333333335e-06, "num_tokens": 3600729.0, "completions/mean_length": 106.375, "completions/min_length": 39.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 48.42857360839844, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.7859811782836914, "rewards/meter/std": 0.34156039357185364, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8739618062973022, "rewards/repeat_soft/std": 0.126714825630188, "rewards/judge_quality/mean": 0.21250000596046448, "rewards/judge_quality/std": 0.14270348846912384, "rewards/total_composite/mean": 0.39180296659469604, "rewards/total_composite/std": 0.17656150460243225, "reward": 0.39180296659469604, "reward_std": 0.17656150460243225, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06401461362838745, "sampling/sampling_logp_difference/max": 1.0067353248596191, "sampling/importance_sampling_ratio/min": 0.5014667510986328, "sampling/importance_sampling_ratio/mean": 1.012511968612671, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4097742661833763, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.06148374988697469, "clip_ratio/high_max": 0.06148374988697469, "clip_ratio/region_mean": 0.06148374988697469, "reward_total_mean": 0.39180296659469604, "reward_meter_mean": 0.7859811782836914, "reward_meter_std": 0.34156039357185364, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8739618062973022, "reward_repeat_soft_std": 0.126714825630188, "reward_judge_quality_mean": 0.21250000596046448, "reward_judge_quality_std": 0.14270348846912384, "reward_total_composite_mean": 0.39180296659469604, "reward_total_composite_std": 0.17656150460243225} {"timestamp_utc": "2026-04-13T12:16:48Z", "mode": "train", "global_step": 2004, "epoch": 0.201305876443998, "loss": 0.0901, "grad_norm": 21.47540855407715, "learning_rate": 3.930303030303031e-06, "num_tokens": 3602137.0, "completions/mean_length": 30.0, "completions/min_length": 25.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.0, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.9228441715240479, "rewards/meter/std": 0.07131259888410568, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9653183221817017, "rewards/repeat_soft/std": 0.03552493825554848, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.43477505445480347, "rewards/total_composite/std": 0.009522650390863419, "reward": 0.43477505445480347, "reward_std": 0.00952264852821827, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10566788166761398, "sampling/sampling_logp_difference/max": 1.7538261413574219, "sampling/importance_sampling_ratio/min": 0.17311032116413116, "sampling/importance_sampling_ratio/mean": 1.012361764907837, "sampling/importance_sampling_ratio/max": 1.6949011087417603, "entropy": 0.5947912894189358, "clip_ratio/low_mean": 0.051282052882015705, "clip_ratio/low_min": 0.051282052882015705, "clip_ratio/high_mean": 0.08320492599159479, "clip_ratio/high_max": 0.08320492599159479, "clip_ratio/region_mean": 0.1344869788736105, "reward_total_mean": 0.43477505445480347, "reward_meter_mean": 0.9228441715240479, "reward_meter_std": 0.07131259888410568, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9653183221817017, "reward_repeat_soft_std": 0.03552493825554848, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.43477505445480347, "reward_total_composite_std": 0.009522650390863419} {"timestamp_utc": "2026-04-13T12:16:56Z", "mode": "train", "global_step": 2005, "epoch": 0.20140632847815168, "loss": -0.0006, "grad_norm": 5.459510803222656, "learning_rate": 3.927272727272727e-06, "num_tokens": 3604963.0, "completions/mean_length": 166.25, "completions/min_length": 137.0, "completions/max_length": 204.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 166.25, "completions/min_terminated_length": 137.0, "completions/max_terminated_length": 204.0, "rewards/meter/mean": 0.9653419852256775, "rewards/meter/std": 0.03195981681346893, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7965164184570312, "rewards/repeat_soft/std": 0.05410079285502434, "rewards/judge_quality/mean": 0.14249999821186066, "rewards/judge_quality/std": 0.031052954494953156, "rewards/total_composite/mean": 0.4087391495704651, "rewards/total_composite/std": 0.019513331353664398, "reward": 0.4087391495704651, "reward_std": 0.01951332949101925, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09489702433347702, "sampling/sampling_logp_difference/max": 1.434617519378662, "sampling/importance_sampling_ratio/min": 0.23820646107196808, "sampling/importance_sampling_ratio/mean": 1.0066559314727783, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6807519569993019, "clip_ratio/low_mean": 0.057409160770475864, "clip_ratio/low_min": 0.057409160770475864, "clip_ratio/high_mean": 0.04812806844711304, "clip_ratio/high_max": 0.04812806844711304, "clip_ratio/region_mean": 0.1055372292175889, "reward_total_mean": 0.4087391495704651, "reward_meter_mean": 0.9653419852256775, "reward_meter_std": 0.03195981681346893, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7965164184570312, "reward_repeat_soft_std": 0.05410079285502434, "reward_judge_quality_mean": 0.14249999821186066, "reward_judge_quality_std": 0.031052954494953156, "reward_total_composite_mean": 0.4087391495704651, "reward_total_composite_std": 0.019513331353664398} {"timestamp_utc": "2026-04-13T12:17:04Z", "mode": "train", "global_step": 2006, "epoch": 0.20150678051230536, "loss": -0.022, "grad_norm": 7.954293251037598, "learning_rate": 3.9242424242424244e-06, "num_tokens": 3607328.0, "completions/mean_length": 104.625, "completions/min_length": 92.0, "completions/max_length": 122.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 104.625, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 122.0, "rewards/meter/mean": 0.9748185873031616, "rewards/meter/std": 0.014069481752812862, "rewards/count_adherence/mean": 0.8250000476837158, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8468031883239746, "rewards/repeat_soft/std": 0.06200995296239853, "rewards/judge_quality/mean": 0.17250001430511475, "rewards/judge_quality/std": 0.02121320739388466, "rewards/total_composite/mean": 0.4012604355812073, "rewards/total_composite/std": 0.027659742161631584, "reward": 0.4012604355812073, "reward_std": 0.027659747749567032, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08801953494548798, "sampling/sampling_logp_difference/max": 2.205092191696167, "sampling/importance_sampling_ratio/min": 0.11024036258459091, "sampling/importance_sampling_ratio/mean": 1.0054447650909424, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.43186598643660545, "clip_ratio/low_mean": 0.04139109933748841, "clip_ratio/low_min": 0.04139109933748841, "clip_ratio/high_mean": 0.043632741551846266, "clip_ratio/high_max": 0.043632741551846266, "clip_ratio/region_mean": 0.08502384088933468, "reward_total_mean": 0.4012604355812073, "reward_meter_mean": 0.9748185873031616, "reward_meter_std": 0.014069481752812862, "reward_count_adherence_mean": 0.8250000476837158, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8468031883239746, "reward_repeat_soft_std": 0.06200995296239853, "reward_judge_quality_mean": 0.17250001430511475, "reward_judge_quality_std": 0.02121320739388466, "reward_total_composite_mean": 0.4012604355812073, "reward_total_composite_std": 0.027659742161631584} {"timestamp_utc": "2026-04-13T12:17:12Z", "mode": "train", "global_step": 2007, "epoch": 0.20160723254645907, "loss": 0.0559, "grad_norm": 6.958494186401367, "learning_rate": 3.921212121212122e-06, "num_tokens": 3609830.0, "completions/mean_length": 109.75, "completions/min_length": 87.0, "completions/max_length": 128.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 109.75, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 128.0, "rewards/meter/mean": 0.9617154002189636, "rewards/meter/std": 0.04410649091005325, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8770691156387329, "rewards/repeat_soft/std": 0.04067332670092583, "rewards/judge_quality/mean": 0.1537500023841858, "rewards/judge_quality/std": 0.01922610029578209, "rewards/total_composite/mean": 0.38770896196365356, "rewards/total_composite/std": 0.015122607350349426, "reward": 0.38770896196365356, "reward_std": 0.015122603625059128, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09767697006464005, "sampling/sampling_logp_difference/max": 1.8629543781280518, "sampling/importance_sampling_ratio/min": 0.15521340072155, "sampling/importance_sampling_ratio/mean": 1.0098620653152466, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6366865932941437, "clip_ratio/low_mean": 0.04750001663342118, "clip_ratio/low_min": 0.04750001663342118, "clip_ratio/high_mean": 0.04482072498649359, "clip_ratio/high_max": 0.04482072498649359, "clip_ratio/region_mean": 0.09232074161991477, "reward_total_mean": 0.38770896196365356, "reward_meter_mean": 0.9617154002189636, "reward_meter_std": 0.04410649091005325, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8770691156387329, "reward_repeat_soft_std": 0.04067332670092583, "reward_judge_quality_mean": 0.1537500023841858, "reward_judge_quality_std": 0.01922610029578209, "reward_total_composite_mean": 0.38770896196365356, "reward_total_composite_std": 0.015122607350349426} {"timestamp_utc": "2026-04-13T12:17:23Z", "mode": "train", "global_step": 2008, "epoch": 0.20170768458061275, "loss": -0.1108, "grad_norm": 2.0068905353546143, "learning_rate": 3.918181818181819e-06, "num_tokens": 3611559.0, "completions/mean_length": 97.125, "completions/min_length": 34.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 37.85714340209961, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.8242834210395813, "rewards/meter/std": 0.3362806439399719, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9566799998283386, "rewards/repeat_soft/std": 0.03871471434831619, "rewards/judge_quality/mean": 0.22750000655651093, "rewards/judge_quality/std": 0.13593590259552002, "rewards/total_composite/mean": 0.43579307198524475, "rewards/total_composite/std": 0.18939152359962463, "reward": 0.43579307198524475, "reward_std": 0.18939152359962463, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14889979362487793, "sampling/sampling_logp_difference/max": 2.3028697967529297, "sampling/importance_sampling_ratio/min": 0.09997153282165527, "sampling/importance_sampling_ratio/mean": 1.0114301443099976, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6406805738806725, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.10330542549490929, "clip_ratio/high_max": 0.10330542549490929, "clip_ratio/region_mean": 0.10330542549490929, "reward_total_mean": 0.43579307198524475, "reward_meter_mean": 0.8242834210395813, "reward_meter_std": 0.3362806439399719, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9566799998283386, "reward_repeat_soft_std": 0.03871471434831619, "reward_judge_quality_mean": 0.22750000655651093, "reward_judge_quality_std": 0.13593590259552002, "reward_total_composite_mean": 0.43579307198524475, "reward_total_composite_std": 0.18939152359962463} {"timestamp_utc": "2026-04-13T12:17:29Z", "mode": "train", "global_step": 2009, "epoch": 0.20180813661476646, "loss": 0.0056, "grad_norm": 9.960348129272461, "learning_rate": 3.915151515151515e-06, "num_tokens": 3613223.0, "completions/mean_length": 35.0, "completions/min_length": 33.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.0, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.7716435194015503, "rewards/meter/std": 0.1756240278482437, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8687934279441833, "rewards/repeat_soft/std": 0.06876115500926971, "rewards/judge_quality/mean": 0.1537500023841858, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.4074971079826355, "rewards/total_composite/std": 0.02326873503625393, "reward": 0.4074971079826355, "reward_std": 0.02326873503625393, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11916100978851318, "sampling/sampling_logp_difference/max": 1.2213225364685059, "sampling/importance_sampling_ratio/min": 0.2948399782180786, "sampling/importance_sampling_ratio/mean": 0.9958372116088867, "sampling/importance_sampling_ratio/max": 1.6207143068313599, "entropy": 0.8184601366519928, "clip_ratio/low_mean": 0.039572364650666714, "clip_ratio/low_min": 0.039572364650666714, "clip_ratio/high_mean": 0.08438057219609618, "clip_ratio/high_max": 0.08438057219609618, "clip_ratio/region_mean": 0.1239529368467629, "reward_total_mean": 0.4074971079826355, "reward_meter_mean": 0.7716435194015503, "reward_meter_std": 0.1756240278482437, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8687934279441833, "reward_repeat_soft_std": 0.06876115500926971, "reward_judge_quality_mean": 0.1537500023841858, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.4074971079826355, "reward_total_composite_std": 0.02326873503625393} {"timestamp_utc": "2026-04-13T12:17:37Z", "mode": "train", "global_step": 2010, "epoch": 0.20190858864892014, "loss": 0.0437, "grad_norm": 6.988424301147461, "learning_rate": 3.912121212121213e-06, "num_tokens": 3615602.0, "completions/mean_length": 119.375, "completions/min_length": 105.0, "completions/max_length": 130.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 119.375, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 130.0, "rewards/meter/mean": 0.9608349204063416, "rewards/meter/std": 0.051238901913166046, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9251853823661804, "rewards/repeat_soft/std": 0.04096551612019539, "rewards/judge_quality/mean": 0.11999999731779099, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4137229025363922, "rewards/total_composite/std": 0.007810365874320269, "reward": 0.4137229025363922, "reward_std": 0.00781036913394928, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09610017389059067, "sampling/sampling_logp_difference/max": 2.1673481464385986, "sampling/importance_sampling_ratio/min": 0.11448080092668533, "sampling/importance_sampling_ratio/mean": 1.0091322660446167, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5776182152330875, "clip_ratio/low_mean": 0.018450709991157055, "clip_ratio/low_min": 0.018450709991157055, "clip_ratio/high_mean": 0.0501365577802062, "clip_ratio/high_max": 0.0501365577802062, "clip_ratio/region_mean": 0.06858726777136326, "reward_total_mean": 0.4137229025363922, "reward_meter_mean": 0.9608349204063416, "reward_meter_std": 0.051238901913166046, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9251853823661804, "reward_repeat_soft_std": 0.04096551612019539, "reward_judge_quality_mean": 0.11999999731779099, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4137229025363922, "reward_total_composite_std": 0.007810365874320269} {"timestamp_utc": "2026-04-13T12:17:43Z", "mode": "train", "global_step": 2011, "epoch": 0.20200904068307382, "loss": 0.1012, "grad_norm": 15.545039176940918, "learning_rate": 3.90909090909091e-06, "num_tokens": 3617019.0, "completions/mean_length": 21.125, "completions/min_length": 18.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.125, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.6375752091407776, "rewards/meter/std": 0.41336917877197266, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.40653860569000244, "rewards/total_composite/std": 0.04030349850654602, "reward": 0.40653860569000244, "reward_std": 0.04030349850654602, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08633841574192047, "sampling/sampling_logp_difference/max": 1.3101986646652222, "sampling/importance_sampling_ratio/min": 0.2697664499282837, "sampling/importance_sampling_ratio/mean": 0.9990162253379822, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4785655103623867, "clip_ratio/low_mean": 0.027083334047347307, "clip_ratio/low_min": 0.027083334047347307, "clip_ratio/high_mean": 0.04612573143094778, "clip_ratio/high_max": 0.04612573143094778, "clip_ratio/region_mean": 0.07320906547829509, "reward_total_mean": 0.40653860569000244, "reward_meter_mean": 0.6375752091407776, "reward_meter_std": 0.41336917877197266, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.40653860569000244, "reward_total_composite_std": 0.04030349850654602} {"timestamp_utc": "2026-04-13T12:17:50Z", "mode": "train", "global_step": 2012, "epoch": 0.20210949271722753, "loss": 0.0113, "grad_norm": 8.610708236694336, "learning_rate": 3.906060606060606e-06, "num_tokens": 3619079.0, "completions/mean_length": 94.5, "completions/min_length": 69.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.5, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.6477097272872925, "rewards/meter/std": 0.3077602982521057, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9543507099151611, "rewards/repeat_soft/std": 0.0374886617064476, "rewards/judge_quality/mean": 0.17625001072883606, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.4186341166496277, "rewards/total_composite/std": 0.035938192158937454, "reward": 0.4186341166496277, "reward_std": 0.035938192158937454, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.127189040184021, "sampling/sampling_logp_difference/max": 2.5745933055877686, "sampling/importance_sampling_ratio/min": 0.07618480175733566, "sampling/importance_sampling_ratio/mean": 1.0038552284240723, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4742015674710274, "clip_ratio/low_mean": 0.05968614085577428, "clip_ratio/low_min": 0.05968614085577428, "clip_ratio/high_mean": 0.06563483458012342, "clip_ratio/high_max": 0.06563483458012342, "clip_ratio/region_mean": 0.1253209754358977, "reward_total_mean": 0.4186341166496277, "reward_meter_mean": 0.6477097272872925, "reward_meter_std": 0.3077602982521057, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9543507099151611, "reward_repeat_soft_std": 0.0374886617064476, "reward_judge_quality_mean": 0.17625001072883606, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.4186341166496277, "reward_total_composite_std": 0.035938192158937454} {"timestamp_utc": "2026-04-13T12:17:55Z", "mode": "train", "global_step": 2013, "epoch": 0.2022099447513812, "loss": 0.0202, "grad_norm": 13.283822059631348, "learning_rate": 3.9030303030303035e-06, "num_tokens": 3620447.0, "completions/mean_length": 16.0, "completions/min_length": 15.0, "completions/max_length": 17.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 16.0, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 17.0, "rewards/meter/mean": 0.9972677230834961, "rewards/meter/std": 0.0004921433283016086, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8947007656097412, "rewards/repeat_soft/std": 0.0385967381298542, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.43143871426582336, "rewards/total_composite/std": 0.005836715456098318, "reward": 0.43143871426582336, "reward_std": 0.005836710799485445, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.051268719136714935, "sampling/sampling_logp_difference/max": 1.1296977996826172, "sampling/importance_sampling_ratio/min": 0.32313090562820435, "sampling/importance_sampling_ratio/mean": 1.0239907503128052, "sampling/importance_sampling_ratio/max": 1.2789902687072754, "entropy": 0.35202306509017944, "clip_ratio/low_mean": 0.007352941203862429, "clip_ratio/low_min": 0.007352941203862429, "clip_ratio/high_mean": 0.023958333767950535, "clip_ratio/high_max": 0.023958333767950535, "clip_ratio/region_mean": 0.031311274971812963, "reward_total_mean": 0.43143871426582336, "reward_meter_mean": 0.9972677230834961, "reward_meter_std": 0.0004921433283016086, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8947007656097412, "reward_repeat_soft_std": 0.0385967381298542, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.43143871426582336, "reward_total_composite_std": 0.005836715456098318} {"timestamp_utc": "2026-04-13T12:18:02Z", "mode": "train", "global_step": 2014, "epoch": 0.20231039678553492, "loss": 0.0049, "grad_norm": 13.62286376953125, "learning_rate": 3.900000000000001e-06, "num_tokens": 3621798.0, "completions/mean_length": 25.875, "completions/min_length": 23.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.875, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.8583083152770996, "rewards/meter/std": 0.34245607256889343, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9488004446029663, "rewards/repeat_soft/std": 0.03874802216887474, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.42600512504577637, "rewards/total_composite/std": 0.03307943791151047, "reward": 0.42600512504577637, "reward_std": 0.03307943791151047, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14248394966125488, "sampling/sampling_logp_difference/max": 1.1403138637542725, "sampling/importance_sampling_ratio/min": 0.31971868872642517, "sampling/importance_sampling_ratio/mean": 1.033398151397705, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0974683165550232, "clip_ratio/low_mean": 0.033518518321216106, "clip_ratio/low_min": 0.033518518321216106, "clip_ratio/high_mean": 0.10463297180831432, "clip_ratio/high_max": 0.10463297180831432, "clip_ratio/region_mean": 0.13815149012953043, "reward_total_mean": 0.42600512504577637, "reward_meter_mean": 0.8583083152770996, "reward_meter_std": 0.34245607256889343, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9488004446029663, "reward_repeat_soft_std": 0.03874802216887474, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.42600512504577637, "reward_total_composite_std": 0.03307943791151047} {"timestamp_utc": "2026-04-13T12:18:14Z", "mode": "train", "global_step": 2015, "epoch": 0.2024108488196886, "loss": -0.1293, "grad_norm": 1.7553294897079468, "learning_rate": 3.896969696969697e-06, "num_tokens": 3623548.0, "completions/mean_length": 101.75, "completions/min_length": 34.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 43.142860412597656, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.8746063113212585, "rewards/meter/std": 0.2304375022649765, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9749953150749207, "rewards/repeat_soft/std": 0.01606718637049198, "rewards/judge_quality/mean": 0.13750000298023224, "rewards/judge_quality/std": 0.0353553406894207, "rewards/total_composite/mean": 0.3840146064758301, "rewards/total_composite/std": 0.1552589386701584, "reward": 0.3840146064758301, "reward_std": 0.1552589386701584, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10318049788475037, "sampling/sampling_logp_difference/max": 1.5672869682312012, "sampling/importance_sampling_ratio/min": 0.20861037075519562, "sampling/importance_sampling_ratio/mean": 1.0245134830474854, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5291539952158928, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.07312493887729943, "clip_ratio/high_max": 0.07312493887729943, "clip_ratio/region_mean": 0.07312493887729943, "reward_total_mean": 0.3840146064758301, "reward_meter_mean": 0.8746063113212585, "reward_meter_std": 0.2304375022649765, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9749953150749207, "reward_repeat_soft_std": 0.01606718637049198, "reward_judge_quality_mean": 0.13750000298023224, "reward_judge_quality_std": 0.0353553406894207, "reward_total_composite_mean": 0.3840146064758301, "reward_total_composite_std": 0.1552589386701584} {"timestamp_utc": "2026-04-13T12:18:20Z", "mode": "train", "global_step": 2016, "epoch": 0.20251130085384228, "loss": 0.0261, "grad_norm": 13.447662353515625, "learning_rate": 3.8939393939393944e-06, "num_tokens": 3625177.0, "completions/mean_length": 34.625, "completions/min_length": 31.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.625, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.9722774028778076, "rewards/meter/std": 0.04150884225964546, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8981412649154663, "rewards/repeat_soft/std": 0.06752867996692657, "rewards/judge_quality/mean": 0.1537500023841858, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.4319431781768799, "rewards/total_composite/std": 0.014848713763058186, "reward": 0.4319431781768799, "reward_std": 0.014848712831735611, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10293879359960556, "sampling/sampling_logp_difference/max": 1.0293388366699219, "sampling/importance_sampling_ratio/min": 0.3572430908679962, "sampling/importance_sampling_ratio/mean": 0.9976475834846497, "sampling/importance_sampling_ratio/max": 1.748008131980896, "entropy": 0.7245897948741913, "clip_ratio/low_mean": 0.024854814866557717, "clip_ratio/low_min": 0.024854814866557717, "clip_ratio/high_mean": 0.05339118931442499, "clip_ratio/high_max": 0.05339118931442499, "clip_ratio/region_mean": 0.07824600418098271, "reward_total_mean": 0.4319431781768799, "reward_meter_mean": 0.9722774028778076, "reward_meter_std": 0.04150884225964546, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8981412649154663, "reward_repeat_soft_std": 0.06752867996692657, "reward_judge_quality_mean": 0.1537500023841858, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.4319431781768799, "reward_total_composite_std": 0.014848713763058186} {"timestamp_utc": "2026-04-13T12:18:27Z", "mode": "train", "global_step": 2017, "epoch": 0.202611752887996, "loss": -0.0535, "grad_norm": 8.292669296264648, "learning_rate": 3.890909090909092e-06, "num_tokens": 3627474.0, "completions/mean_length": 108.125, "completions/min_length": 92.0, "completions/max_length": 121.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.125, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.8273603916168213, "rewards/meter/std": 0.2854452133178711, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8415908217430115, "rewards/repeat_soft/std": 0.04468485340476036, "rewards/judge_quality/mean": 0.17250001430511475, "rewards/judge_quality/std": 0.02121320739388466, "rewards/total_composite/mean": 0.4153048098087311, "rewards/total_composite/std": 0.04054592177271843, "reward": 0.4153048098087311, "reward_std": 0.04054592177271843, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1157137006521225, "sampling/sampling_logp_difference/max": 2.0424206256866455, "sampling/importance_sampling_ratio/min": 0.12971433997154236, "sampling/importance_sampling_ratio/mean": 1.0190651416778564, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6621493026614189, "clip_ratio/low_mean": 0.04455561004579067, "clip_ratio/low_min": 0.04455561004579067, "clip_ratio/high_mean": 0.06098366668447852, "clip_ratio/high_max": 0.06098366668447852, "clip_ratio/region_mean": 0.1055392767302692, "reward_total_mean": 0.4153048098087311, "reward_meter_mean": 0.8273603916168213, "reward_meter_std": 0.2854452133178711, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8415908217430115, "reward_repeat_soft_std": 0.04468485340476036, "reward_judge_quality_mean": 0.17250001430511475, "reward_judge_quality_std": 0.02121320739388466, "reward_total_composite_mean": 0.4153048098087311, "reward_total_composite_std": 0.04054592177271843} {"timestamp_utc": "2026-04-13T12:18:34Z", "mode": "train", "global_step": 2018, "epoch": 0.20271220492214967, "loss": -0.0132, "grad_norm": 12.649149894714355, "learning_rate": 3.887878787878788e-06, "num_tokens": 3629025.0, "completions/mean_length": 29.875, "completions/min_length": 26.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.875, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9288086891174316, "rewards/meter/std": 0.05646306276321411, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.889045238494873, "rewards/repeat_soft/std": 0.10175919532775879, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4239156246185303, "rewards/total_composite/std": 0.013974902220070362, "reward": 0.4239156246185303, "reward_std": 0.013974908739328384, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1317238211631775, "sampling/sampling_logp_difference/max": 1.3933568000793457, "sampling/importance_sampling_ratio/min": 0.2482406198978424, "sampling/importance_sampling_ratio/mean": 0.9737530946731567, "sampling/importance_sampling_ratio/max": 1.6525967121124268, "entropy": 0.5082592852413654, "clip_ratio/low_mean": 0.03976034792140126, "clip_ratio/low_min": 0.03976034792140126, "clip_ratio/high_mean": 0.11229077726602554, "clip_ratio/high_max": 0.11229077726602554, "clip_ratio/region_mean": 0.1520511251874268, "reward_total_mean": 0.4239156246185303, "reward_meter_mean": 0.9288086891174316, "reward_meter_std": 0.05646306276321411, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.889045238494873, "reward_repeat_soft_std": 0.10175919532775879, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4239156246185303, "reward_total_composite_std": 0.013974902220070362} {"timestamp_utc": "2026-04-13T12:18:45Z", "mode": "train", "global_step": 2019, "epoch": 0.20281265695630338, "loss": -0.0915, "grad_norm": 1.7585482597351074, "learning_rate": 3.884848484848485e-06, "num_tokens": 3630678.0, "completions/mean_length": 93.625, "completions/min_length": 31.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 33.85714340209961, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.722576379776001, "rewards/meter/std": 0.35335662961006165, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9477214813232422, "rewards/repeat_soft/std": 0.039741020649671555, "rewards/judge_quality/mean": 0.17875000834465027, "rewards/judge_quality/std": 0.1056189239025116, "rewards/total_composite/mean": 0.38987091183662415, "rewards/total_composite/std": 0.1731443852186203, "reward": 0.38987091183662415, "reward_std": 0.1731443852186203, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1067752093076706, "sampling/sampling_logp_difference/max": 1.1385347843170166, "sampling/importance_sampling_ratio/min": 0.3202879726886749, "sampling/importance_sampling_ratio/mean": 1.000455379486084, "sampling/importance_sampling_ratio/max": 1.959191918373108, "entropy": 0.5118012577295303, "clip_ratio/low_mean": 0.007352941203862429, "clip_ratio/low_min": 0.007352941203862429, "clip_ratio/high_mean": 0.06762545485980809, "clip_ratio/high_max": 0.06762545485980809, "clip_ratio/region_mean": 0.07497839606367052, "reward_total_mean": 0.38987091183662415, "reward_meter_mean": 0.722576379776001, "reward_meter_std": 0.35335662961006165, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9477214813232422, "reward_repeat_soft_std": 0.039741020649671555, "reward_judge_quality_mean": 0.17875000834465027, "reward_judge_quality_std": 0.1056189239025116, "reward_total_composite_mean": 0.38987091183662415, "reward_total_composite_std": 0.1731443852186203} {"timestamp_utc": "2026-04-13T12:18:52Z", "mode": "train", "global_step": 2020, "epoch": 0.20291310899045706, "loss": 0.0009, "grad_norm": 8.50749397277832, "learning_rate": 3.881818181818182e-06, "num_tokens": 3632791.0, "completions/mean_length": 82.125, "completions/min_length": 59.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.125, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.9119631052017212, "rewards/meter/std": 0.2121911197900772, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8878892660140991, "rewards/repeat_soft/std": 0.0242925938218832, "rewards/judge_quality/mean": 0.17625001072883606, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.4374694526195526, "rewards/total_composite/std": 0.026745209470391273, "reward": 0.4374694526195526, "reward_std": 0.026745207607746124, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11416690796613693, "sampling/sampling_logp_difference/max": 1.469538688659668, "sampling/importance_sampling_ratio/min": 0.2300315797328949, "sampling/importance_sampling_ratio/mean": 1.0092949867248535, "sampling/importance_sampling_ratio/max": 1.9388819932937622, "entropy": 0.7308895066380501, "clip_ratio/low_mean": 0.033150406554341316, "clip_ratio/low_min": 0.033150406554341316, "clip_ratio/high_mean": 0.09055669419467449, "clip_ratio/high_max": 0.09055669419467449, "clip_ratio/region_mean": 0.12370710074901581, "reward_total_mean": 0.4374694526195526, "reward_meter_mean": 0.9119631052017212, "reward_meter_std": 0.2121911197900772, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8878892660140991, "reward_repeat_soft_std": 0.0242925938218832, "reward_judge_quality_mean": 0.17625001072883606, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.4374694526195526, "reward_total_composite_std": 0.026745209470391273} {"timestamp_utc": "2026-04-13T12:19:00Z", "mode": "train", "global_step": 2021, "epoch": 0.20301356102461074, "loss": 0.0381, "grad_norm": 8.613554000854492, "learning_rate": 3.878787878787879e-06, "num_tokens": 3634310.0, "completions/mean_length": 47.875, "completions/min_length": 33.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.875, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.6935790777206421, "rewards/meter/std": 0.365664541721344, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9693573713302612, "rewards/repeat_soft/std": 0.012484983541071415, "rewards/judge_quality/mean": 0.1574999988079071, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.41753631830215454, "rewards/total_composite/std": 0.039422620087862015, "reward": 0.41753631830215454, "reward_std": 0.03942260518670082, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09699514508247375, "sampling/sampling_logp_difference/max": 2.03543758392334, "sampling/importance_sampling_ratio/min": 0.13062331080436707, "sampling/importance_sampling_ratio/mean": 1.0070172548294067, "sampling/importance_sampling_ratio/max": 1.9344388246536255, "entropy": 0.5120501779019833, "clip_ratio/low_mean": 0.03184592956677079, "clip_ratio/low_min": 0.03184592956677079, "clip_ratio/high_mean": 0.06646964652463794, "clip_ratio/high_max": 0.06646964652463794, "clip_ratio/region_mean": 0.09831557609140873, "reward_total_mean": 0.41753631830215454, "reward_meter_mean": 0.6935790777206421, "reward_meter_std": 0.365664541721344, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9693573713302612, "reward_repeat_soft_std": 0.012484983541071415, "reward_judge_quality_mean": 0.1574999988079071, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.41753631830215454, "reward_total_composite_std": 0.039422620087862015} {"timestamp_utc": "2026-04-13T12:19:11Z", "mode": "train", "global_step": 2022, "epoch": 0.20311401305876445, "loss": -0.2056, "grad_norm": 1.7340670824050903, "learning_rate": 3.875757575757576e-06, "num_tokens": 3636463.0, "completions/mean_length": 157.125, "completions/min_length": 93.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 106.42857360839844, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.811650276184082, "rewards/meter/std": 0.2834619879722595, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8670904040336609, "rewards/repeat_soft/std": 0.04739189147949219, "rewards/judge_quality/mean": 0.1262499988079071, "rewards/judge_quality/std": 0.04103570431470871, "rewards/total_composite/mean": 0.35624122619628906, "rewards/total_composite/std": 0.14686189591884613, "reward": 0.35624122619628906, "reward_std": 0.14686188101768494, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1068727895617485, "sampling/sampling_logp_difference/max": 3.1300125122070312, "sampling/importance_sampling_ratio/min": 0.04371724650263786, "sampling/importance_sampling_ratio/mean": 1.0075139999389648, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4911436326801777, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09701418410986662, "clip_ratio/high_max": 0.09701418410986662, "clip_ratio/region_mean": 0.09701418410986662, "reward_total_mean": 0.35624122619628906, "reward_meter_mean": 0.811650276184082, "reward_meter_std": 0.2834619879722595, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8670904040336609, "reward_repeat_soft_std": 0.04739189147949219, "reward_judge_quality_mean": 0.1262499988079071, "reward_judge_quality_std": 0.04103570431470871, "reward_total_composite_mean": 0.35624122619628906, "reward_total_composite_std": 0.14686189591884613} {"timestamp_utc": "2026-04-13T12:19:18Z", "mode": "train", "global_step": 2023, "epoch": 0.20321446509291813, "loss": 0.064, "grad_norm": 10.871502876281738, "learning_rate": 3.872727272727273e-06, "num_tokens": 3638197.0, "completions/mean_length": 51.75, "completions/min_length": 45.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.75, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9767289161682129, "rewards/meter/std": 0.020327460020780563, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9791615009307861, "rewards/repeat_soft/std": 0.02554360218346119, "rewards/judge_quality/mean": 0.24625001847743988, "rewards/judge_quality/std": 0.2722361087799072, "rewards/total_composite/mean": 0.5042101144790649, "rewards/total_composite/std": 0.1774170845746994, "reward": 0.5042101144790649, "reward_std": 0.1774170994758606, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12130969017744064, "sampling/sampling_logp_difference/max": 1.0781450271606445, "sampling/importance_sampling_ratio/min": 0.34022602438926697, "sampling/importance_sampling_ratio/mean": 1.0086771249771118, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7564574480056763, "clip_ratio/low_mean": 0.09432055428624153, "clip_ratio/low_min": 0.09432055428624153, "clip_ratio/high_mean": 0.010869565419852734, "clip_ratio/high_max": 0.010869565419852734, "clip_ratio/region_mean": 0.10519011970609426, "reward_total_mean": 0.5042101144790649, "reward_meter_mean": 0.9767289161682129, "reward_meter_std": 0.020327460020780563, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9791615009307861, "reward_repeat_soft_std": 0.02554360218346119, "reward_judge_quality_mean": 0.24625001847743988, "reward_judge_quality_std": 0.2722361087799072, "reward_total_composite_mean": 0.5042101144790649, "reward_total_composite_std": 0.1774170845746994} {"timestamp_utc": "2026-04-13T12:19:24Z", "mode": "train", "global_step": 2024, "epoch": 0.20331491712707184, "loss": 0.0081, "grad_norm": 14.9652099609375, "learning_rate": 3.86969696969697e-06, "num_tokens": 3639790.0, "completions/mean_length": 36.125, "completions/min_length": 32.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.125, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.8460080623626709, "rewards/meter/std": 0.3138962984085083, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9143238663673401, "rewards/repeat_soft/std": 0.0503779835999012, "rewards/judge_quality/mean": 0.13875001668930054, "rewards/judge_quality/std": 0.015526479110121727, "rewards/total_composite/mean": 0.4125252068042755, "rewards/total_composite/std": 0.030189499258995056, "reward": 0.4125252068042755, "reward_std": 0.030189495533704758, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05745663493871689, "sampling/sampling_logp_difference/max": 1.122283935546875, "sampling/importance_sampling_ratio/min": 0.3255354166030884, "sampling/importance_sampling_ratio/mean": 0.9936563968658447, "sampling/importance_sampling_ratio/max": 1.621734619140625, "entropy": 0.26591492630541325, "clip_ratio/low_mean": 0.017330858390778303, "clip_ratio/low_min": 0.017330858390778303, "clip_ratio/high_mean": 0.021464647026732564, "clip_ratio/high_max": 0.021464647026732564, "clip_ratio/region_mean": 0.03879550541751087, "reward_total_mean": 0.4125252068042755, "reward_meter_mean": 0.8460080623626709, "reward_meter_std": 0.3138962984085083, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9143238663673401, "reward_repeat_soft_std": 0.0503779835999012, "reward_judge_quality_mean": 0.13875001668930054, "reward_judge_quality_std": 0.015526479110121727, "reward_total_composite_mean": 0.4125252068042755, "reward_total_composite_std": 0.030189499258995056} {"timestamp_utc": "2026-04-13T12:19:30Z", "mode": "train", "global_step": 2025, "epoch": 0.20341536916122552, "loss": 0.0021, "grad_norm": 10.698853492736816, "learning_rate": 3.866666666666667e-06, "num_tokens": 3641540.0, "completions/mean_length": 51.75, "completions/min_length": 43.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.75, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.6679095029830933, "rewards/meter/std": 0.29510411620140076, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9887790679931641, "rewards/repeat_soft/std": 0.016280222684144974, "rewards/judge_quality/mean": 0.1537500023841858, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.4143340587615967, "rewards/total_composite/std": 0.02845674753189087, "reward": 0.4143340587615967, "reward_std": 0.028456738218665123, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1274341642856598, "sampling/sampling_logp_difference/max": 1.6919660568237305, "sampling/importance_sampling_ratio/min": 0.1841571182012558, "sampling/importance_sampling_ratio/mean": 1.0127040147781372, "sampling/importance_sampling_ratio/max": 1.8242231607437134, "entropy": 0.6411586254835129, "clip_ratio/low_mean": 0.04926663590595126, "clip_ratio/low_min": 0.04926663590595126, "clip_ratio/high_mean": 0.05039568850770593, "clip_ratio/high_max": 0.05039568850770593, "clip_ratio/region_mean": 0.09966232441365719, "reward_total_mean": 0.4143340587615967, "reward_meter_mean": 0.6679095029830933, "reward_meter_std": 0.29510411620140076, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9887790679931641, "reward_repeat_soft_std": 0.016280222684144974, "reward_judge_quality_mean": 0.1537500023841858, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.4143340587615967, "reward_total_composite_std": 0.02845674753189087} {"timestamp_utc": "2026-04-13T12:19:37Z", "mode": "train", "global_step": 2026, "epoch": 0.2035158211953792, "loss": -0.0773, "grad_norm": 11.50498104095459, "learning_rate": 3.863636363636364e-06, "num_tokens": 3642999.0, "completions/mean_length": 35.375, "completions/min_length": 26.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.375, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.5798919796943665, "rewards/meter/std": 0.43411311507225037, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9656744003295898, "rewards/repeat_soft/std": 0.04045727103948593, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.40139061212539673, "rewards/total_composite/std": 0.03712591901421547, "reward": 0.40139061212539673, "reward_std": 0.03712591901421547, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12781719863414764, "sampling/sampling_logp_difference/max": 1.2028522491455078, "sampling/importance_sampling_ratio/min": 0.3003363609313965, "sampling/importance_sampling_ratio/mean": 1.0064102411270142, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6434668265283108, "clip_ratio/low_mean": 0.058067780919373035, "clip_ratio/low_min": 0.058067780919373035, "clip_ratio/high_mean": 0.05798420403152704, "clip_ratio/high_max": 0.05798420403152704, "clip_ratio/region_mean": 0.11605198495090008, "reward_total_mean": 0.40139061212539673, "reward_meter_mean": 0.5798919796943665, "reward_meter_std": 0.43411311507225037, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9656744003295898, "reward_repeat_soft_std": 0.04045727103948593, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.40139061212539673, "reward_total_composite_std": 0.03712591901421547} {"timestamp_utc": "2026-04-13T12:19:43Z", "mode": "train", "global_step": 2027, "epoch": 0.2036162732295329, "loss": 0.0414, "grad_norm": 11.639140129089355, "learning_rate": 3.860606060606061e-06, "num_tokens": 3644690.0, "completions/mean_length": 58.375, "completions/min_length": 51.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.375, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.8482519388198853, "rewards/meter/std": 0.30271652340888977, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9650776386260986, "rewards/repeat_soft/std": 0.03526902198791504, "rewards/judge_quality/mean": 0.1612500101327896, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.4340474307537079, "rewards/total_composite/std": 0.030730562284588814, "reward": 0.4340474307537079, "reward_std": 0.030730556696653366, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10567151010036469, "sampling/sampling_logp_difference/max": 1.1417315006256104, "sampling/importance_sampling_ratio/min": 0.319265753030777, "sampling/importance_sampling_ratio/mean": 1.0057364702224731, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.69110918790102, "clip_ratio/low_mean": 0.036418041214346886, "clip_ratio/low_min": 0.036418041214346886, "clip_ratio/high_mean": 0.07654706481844187, "clip_ratio/high_max": 0.07654706481844187, "clip_ratio/region_mean": 0.11296510603278875, "reward_total_mean": 0.4340474307537079, "reward_meter_mean": 0.8482519388198853, "reward_meter_std": 0.30271652340888977, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9650776386260986, "reward_repeat_soft_std": 0.03526902198791504, "reward_judge_quality_mean": 0.1612500101327896, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.4340474307537079, "reward_total_composite_std": 0.030730562284588814} {"timestamp_utc": "2026-04-13T12:19:51Z", "mode": "train", "global_step": 2028, "epoch": 0.2037167252636866, "loss": -0.1285, "grad_norm": 7.573558807373047, "learning_rate": 3.857575757575758e-06, "num_tokens": 3647183.0, "completions/mean_length": 120.625, "completions/min_length": 81.0, "completions/max_length": 171.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 120.625, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 171.0, "rewards/meter/mean": 0.8907461166381836, "rewards/meter/std": 0.12185462564229965, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.1035098284482956, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8501641154289246, "rewards/repeat_soft/std": 0.04492751881480217, "rewards/judge_quality/mean": 0.1574999988079071, "rewards/judge_quality/std": 0.031052954494953156, "rewards/total_composite/mean": 0.3936898708343506, "rewards/total_composite/std": 0.02330118790268898, "reward": 0.3936898708343506, "reward_std": 0.02330118604004383, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10553734749555588, "sampling/sampling_logp_difference/max": 1.3691177368164062, "sampling/importance_sampling_ratio/min": 0.2543312609195709, "sampling/importance_sampling_ratio/mean": 1.0143228769302368, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6704464554786682, "clip_ratio/low_mean": 0.05444397870451212, "clip_ratio/low_min": 0.05444397870451212, "clip_ratio/high_mean": 0.057956038042902946, "clip_ratio/high_max": 0.057956038042902946, "clip_ratio/region_mean": 0.11240001674741507, "reward_total_mean": 0.3936898708343506, "reward_meter_mean": 0.8907461166381836, "reward_meter_std": 0.12185462564229965, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.1035098284482956, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8501641154289246, "reward_repeat_soft_std": 0.04492751881480217, "reward_judge_quality_mean": 0.1574999988079071, "reward_judge_quality_std": 0.031052954494953156, "reward_total_composite_mean": 0.3936898708343506, "reward_total_composite_std": 0.02330118790268898} {"timestamp_utc": "2026-04-13T12:19:59Z", "mode": "train", "global_step": 2029, "epoch": 0.20381717729784027, "loss": -0.1184, "grad_norm": 6.718632698059082, "learning_rate": 3.8545454545454545e-06, "num_tokens": 3649393.0, "completions/mean_length": 118.25, "completions/min_length": 99.0, "completions/max_length": 176.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 118.25, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 176.0, "rewards/meter/mean": 0.7876110672950745, "rewards/meter/std": 0.16306620836257935, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7729731798171997, "rewards/repeat_soft/std": 0.08785118907690048, "rewards/judge_quality/mean": 0.16500000655651093, "rewards/judge_quality/std": 0.02777460776269436, "rewards/total_composite/mean": 0.40055200457572937, "rewards/total_composite/std": 0.027080491185188293, "reward": 0.40055200457572937, "reward_std": 0.027080485597252846, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0912838876247406, "sampling/sampling_logp_difference/max": 2.8005564212799072, "sampling/importance_sampling_ratio/min": 0.0607762336730957, "sampling/importance_sampling_ratio/mean": 1.0105650424957275, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4359697103500366, "clip_ratio/low_mean": 0.0342613086104393, "clip_ratio/low_min": 0.0342613086104393, "clip_ratio/high_mean": 0.05099095404148102, "clip_ratio/high_max": 0.05099095404148102, "clip_ratio/region_mean": 0.08525226265192032, "reward_total_mean": 0.40055200457572937, "reward_meter_mean": 0.7876110672950745, "reward_meter_std": 0.16306620836257935, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7729731798171997, "reward_repeat_soft_std": 0.08785118907690048, "reward_judge_quality_mean": 0.16500000655651093, "reward_judge_quality_std": 0.02777460776269436, "reward_total_composite_mean": 0.40055200457572937, "reward_total_composite_std": 0.027080491185188293} {"timestamp_utc": "2026-04-13T12:20:06Z", "mode": "train", "global_step": 2030, "epoch": 0.20391762933199398, "loss": -0.0294, "grad_norm": 13.128202438354492, "learning_rate": 3.851515151515152e-06, "num_tokens": 3650949.0, "completions/mean_length": 41.5, "completions/min_length": 36.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.5, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.8338658809661865, "rewards/meter/std": 0.3192678391933441, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9496046900749207, "rewards/repeat_soft/std": 0.04710354283452034, "rewards/judge_quality/mean": 0.16500000655651093, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.4313012957572937, "rewards/total_composite/std": 0.03520743176341057, "reward": 0.4313012957572937, "reward_std": 0.03520742803812027, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11196162551641464, "sampling/sampling_logp_difference/max": 1.4962571859359741, "sampling/importance_sampling_ratio/min": 0.22396688163280487, "sampling/importance_sampling_ratio/mean": 1.0043363571166992, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5714318752288818, "clip_ratio/low_mean": 0.03207236900925636, "clip_ratio/low_min": 0.03207236900925636, "clip_ratio/high_mean": 0.08106857095845044, "clip_ratio/high_max": 0.08106857095845044, "clip_ratio/region_mean": 0.1131409399677068, "reward_total_mean": 0.4313012957572937, "reward_meter_mean": 0.8338658809661865, "reward_meter_std": 0.3192678391933441, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9496046900749207, "reward_repeat_soft_std": 0.04710354283452034, "reward_judge_quality_mean": 0.16500000655651093, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.4313012957572937, "reward_total_composite_std": 0.03520743176341057} {"timestamp_utc": "2026-04-13T12:20:12Z", "mode": "train", "global_step": 2031, "epoch": 0.20401808136614766, "loss": 0.0286, "grad_norm": 12.468624114990234, "learning_rate": 3.848484848484848e-06, "num_tokens": 3652423.0, "completions/mean_length": 21.25, "completions/min_length": 17.0, "completions/max_length": 24.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.25, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 24.0, "rewards/meter/mean": 0.6922364830970764, "rewards/meter/std": 0.42275819182395935, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9160686731338501, "rewards/repeat_soft/std": 0.04868149757385254, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4049033522605896, "rewards/total_composite/std": 0.046748362481594086, "reward": 0.4049033522605896, "reward_std": 0.046748362481594086, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10105178505182266, "sampling/sampling_logp_difference/max": 1.5017375946044922, "sampling/importance_sampling_ratio/min": 0.22274281084537506, "sampling/importance_sampling_ratio/mean": 0.9819271564483643, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4016204699873924, "clip_ratio/low_mean": 0.01711956551298499, "clip_ratio/low_min": 0.01711956551298499, "clip_ratio/high_mean": 0.07495747972279787, "clip_ratio/high_max": 0.07495747972279787, "clip_ratio/region_mean": 0.09207704523578286, "reward_total_mean": 0.4049033522605896, "reward_meter_mean": 0.6922364830970764, "reward_meter_std": 0.42275819182395935, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9160686731338501, "reward_repeat_soft_std": 0.04868149757385254, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4049033522605896, "reward_total_composite_std": 0.046748362481594086} {"timestamp_utc": "2026-04-13T12:20:19Z", "mode": "train", "global_step": 2032, "epoch": 0.20411853340030137, "loss": 0.0103, "grad_norm": 23.99254035949707, "learning_rate": 3.8454545454545454e-06, "num_tokens": 3653854.0, "completions/mean_length": 17.875, "completions/min_length": 14.0, "completions/max_length": 22.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 17.875, "completions/min_terminated_length": 14.0, "completions/max_terminated_length": 22.0, "rewards/meter/mean": 0.9383523464202881, "rewards/meter/std": 0.11129981279373169, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9575029015541077, "rewards/repeat_soft/std": 0.014133882708847523, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4351148009300232, "rewards/total_composite/std": 0.01072826236486435, "reward": 0.4351148009300232, "reward_std": 0.010728267952799797, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08517216145992279, "sampling/sampling_logp_difference/max": 1.2159996032714844, "sampling/importance_sampling_ratio/min": 0.2964135706424713, "sampling/importance_sampling_ratio/mean": 1.00950288772583, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.506832730025053, "clip_ratio/low_mean": 0.01666666753590107, "clip_ratio/low_min": 0.01666666753590107, "clip_ratio/high_mean": 0.11809424310922623, "clip_ratio/high_max": 0.11809424310922623, "clip_ratio/region_mean": 0.1347609106451273, "reward_total_mean": 0.4351148009300232, "reward_meter_mean": 0.9383523464202881, "reward_meter_std": 0.11129981279373169, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9575029015541077, "reward_repeat_soft_std": 0.014133882708847523, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4351148009300232, "reward_total_composite_std": 0.01072826236486435} {"timestamp_utc": "2026-04-13T12:20:26Z", "mode": "train", "global_step": 2033, "epoch": 0.20421898543445505, "loss": -0.0539, "grad_norm": 7.2069854736328125, "learning_rate": 3.842424242424243e-06, "num_tokens": 3656268.0, "completions/mean_length": 127.75, "completions/min_length": 101.0, "completions/max_length": 156.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 127.75, "completions/min_terminated_length": 101.0, "completions/max_terminated_length": 156.0, "rewards/meter/mean": 0.9627584218978882, "rewards/meter/std": 0.025766663253307343, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7789667844772339, "rewards/repeat_soft/std": 0.07983645796775818, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.03207135573029518, "rewards/total_composite/mean": 0.41072261333465576, "rewards/total_composite/std": 0.02998325042426586, "reward": 0.41072261333465576, "reward_std": 0.029983246698975563, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08343739062547684, "sampling/sampling_logp_difference/max": 1.908365249633789, "sampling/importance_sampling_ratio/min": 0.1483226716518402, "sampling/importance_sampling_ratio/mean": 1.0080602169036865, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.43094078078866005, "clip_ratio/low_mean": 0.03441926138475537, "clip_ratio/low_min": 0.03441926138475537, "clip_ratio/high_mean": 0.03305764775723219, "clip_ratio/high_max": 0.03305764775723219, "clip_ratio/region_mean": 0.06747690914198756, "reward_total_mean": 0.41072261333465576, "reward_meter_mean": 0.9627584218978882, "reward_meter_std": 0.025766663253307343, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7789667844772339, "reward_repeat_soft_std": 0.07983645796775818, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.03207135573029518, "reward_total_composite_mean": 0.41072261333465576, "reward_total_composite_std": 0.02998325042426586} {"timestamp_utc": "2026-04-13T12:20:33Z", "mode": "train", "global_step": 2034, "epoch": 0.20431943746860873, "loss": 0.0586, "grad_norm": 7.509989261627197, "learning_rate": 3.839393939393939e-06, "num_tokens": 3658045.0, "completions/mean_length": 52.125, "completions/min_length": 41.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.125, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.8884906768798828, "rewards/meter/std": 0.1771586537361145, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8319612741470337, "rewards/repeat_soft/std": 0.06831370294094086, "rewards/judge_quality/mean": 0.14625000953674316, "rewards/judge_quality/std": 0.01922610215842724, "rewards/total_composite/mean": 0.3588881194591522, "rewards/total_composite/std": 0.14733052253723145, "reward": 0.3588881194591522, "reward_std": 0.14733052253723145, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12376967817544937, "sampling/sampling_logp_difference/max": 1.9423317909240723, "sampling/importance_sampling_ratio/min": 0.14336925745010376, "sampling/importance_sampling_ratio/mean": 0.9940099120140076, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5854336768388748, "clip_ratio/low_mean": 0.008333333767950535, "clip_ratio/low_min": 0.008333333767950535, "clip_ratio/high_mean": 0.10637603979557753, "clip_ratio/high_max": 0.10637603979557753, "clip_ratio/region_mean": 0.11470937356352806, "reward_total_mean": 0.3588881194591522, "reward_meter_mean": 0.8884906768798828, "reward_meter_std": 0.1771586537361145, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8319612741470337, "reward_repeat_soft_std": 0.06831370294094086, "reward_judge_quality_mean": 0.14625000953674316, "reward_judge_quality_std": 0.01922610215842724, "reward_total_composite_mean": 0.3588881194591522, "reward_total_composite_std": 0.14733052253723145} {"timestamp_utc": "2026-04-13T12:20:40Z", "mode": "train", "global_step": 2035, "epoch": 0.20441988950276244, "loss": -0.0323, "grad_norm": 7.591353893280029, "learning_rate": 3.836363636363636e-06, "num_tokens": 3660138.0, "completions/mean_length": 74.625, "completions/min_length": 68.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.625, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.9790351986885071, "rewards/meter/std": 0.03450845554471016, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8338398933410645, "rewards/repeat_soft/std": 0.08215414732694626, "rewards/judge_quality/mean": 0.18000000715255737, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4396231174468994, "rewards/total_composite/std": 0.010331184603273869, "reward": 0.4396231174468994, "reward_std": 0.01033118087798357, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08554603159427643, "sampling/sampling_logp_difference/max": 1.9737772941589355, "sampling/importance_sampling_ratio/min": 0.13893108069896698, "sampling/importance_sampling_ratio/mean": 1.0047028064727783, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.428936704993248, "clip_ratio/low_mean": 0.036277403589338064, "clip_ratio/low_min": 0.036277403589338064, "clip_ratio/high_mean": 0.046252098865807056, "clip_ratio/high_max": 0.046252098865807056, "clip_ratio/region_mean": 0.08252950245514512, "reward_total_mean": 0.4396231174468994, "reward_meter_mean": 0.9790351986885071, "reward_meter_std": 0.03450845554471016, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8338398933410645, "reward_repeat_soft_std": 0.08215414732694626, "reward_judge_quality_mean": 0.18000000715255737, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4396231174468994, "reward_total_composite_std": 0.010331184603273869} {"timestamp_utc": "2026-04-13T12:20:47Z", "mode": "train", "global_step": 2036, "epoch": 0.20452034153691612, "loss": 0.0618, "grad_norm": 12.391810417175293, "learning_rate": 3.833333333333334e-06, "num_tokens": 3662473.0, "completions/mean_length": 103.875, "completions/min_length": 89.0, "completions/max_length": 121.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 103.875, "completions/min_terminated_length": 89.0, "completions/max_terminated_length": 121.0, "rewards/meter/mean": 0.746888279914856, "rewards/meter/std": 0.25801926851272583, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8703850507736206, "rewards/repeat_soft/std": 0.024696068838238716, "rewards/judge_quality/mean": 0.16875000298023224, "rewards/judge_quality/std": 0.022320717573165894, "rewards/total_composite/mean": 0.4135923385620117, "rewards/total_composite/std": 0.031244825571775436, "reward": 0.4135923385620117, "reward_std": 0.031244821846485138, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08883036673069, "sampling/sampling_logp_difference/max": 1.893946886062622, "sampling/importance_sampling_ratio/min": 0.15047672390937805, "sampling/importance_sampling_ratio/mean": 1.0045208930969238, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4564935080707073, "clip_ratio/low_mean": 0.03486485220491886, "clip_ratio/low_min": 0.03486485220491886, "clip_ratio/high_mean": 0.04028744110837579, "clip_ratio/high_max": 0.04028744110837579, "clip_ratio/region_mean": 0.07515229331329465, "reward_total_mean": 0.4135923385620117, "reward_meter_mean": 0.746888279914856, "reward_meter_std": 0.25801926851272583, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8703850507736206, "reward_repeat_soft_std": 0.024696068838238716, "reward_judge_quality_mean": 0.16875000298023224, "reward_judge_quality_std": 0.022320717573165894, "reward_total_composite_mean": 0.4135923385620117, "reward_total_composite_std": 0.031244825571775436} {"timestamp_utc": "2026-04-13T12:20:53Z", "mode": "train", "global_step": 2037, "epoch": 0.20462079357106983, "loss": -0.0844, "grad_norm": 14.103687286376953, "learning_rate": 3.830303030303031e-06, "num_tokens": 3664098.0, "completions/mean_length": 37.125, "completions/min_length": 32.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.125, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.8908556699752808, "rewards/meter/std": 0.20262905955314636, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8498884439468384, "rewards/repeat_soft/std": 0.12281721830368042, "rewards/judge_quality/mean": 0.14250001311302185, "rewards/judge_quality/std": 0.013887306675314903, "rewards/total_composite/mean": 0.40948760509490967, "rewards/total_composite/std": 0.019680894911289215, "reward": 0.40948760509490967, "reward_std": 0.019680896773934364, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09558341652154922, "sampling/sampling_logp_difference/max": 1.2192243337631226, "sampling/importance_sampling_ratio/min": 0.2954592704772949, "sampling/importance_sampling_ratio/mean": 1.0035276412963867, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7432537972927094, "clip_ratio/low_mean": 0.048447174951434135, "clip_ratio/low_min": 0.048447174951434135, "clip_ratio/high_mean": 0.038103504572063684, "clip_ratio/high_max": 0.038103504572063684, "clip_ratio/region_mean": 0.08655067952349782, "reward_total_mean": 0.40948760509490967, "reward_meter_mean": 0.8908556699752808, "reward_meter_std": 0.20262905955314636, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8498884439468384, "reward_repeat_soft_std": 0.12281721830368042, "reward_judge_quality_mean": 0.14250001311302185, "reward_judge_quality_std": 0.013887306675314903, "reward_total_composite_mean": 0.40948760509490967, "reward_total_composite_std": 0.019680894911289215} {"timestamp_utc": "2026-04-13T12:21:00Z", "mode": "train", "global_step": 2038, "epoch": 0.2047212456052235, "loss": 0.0111, "grad_norm": 14.108630180358887, "learning_rate": 3.827272727272728e-06, "num_tokens": 3665743.0, "completions/mean_length": 45.625, "completions/min_length": 36.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.625, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.6388722658157349, "rewards/meter/std": 0.42735129594802856, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9902987480163574, "rewards/repeat_soft/std": 0.009319685399532318, "rewards/judge_quality/mean": 0.22500000894069672, "rewards/judge_quality/std": 0.13887301087379456, "rewards/total_composite/mean": 0.45695754885673523, "rewards/total_composite/std": 0.11092603206634521, "reward": 0.45695754885673523, "reward_std": 0.11092603206634521, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11713860929012299, "sampling/sampling_logp_difference/max": 1.2782936096191406, "sampling/importance_sampling_ratio/min": 0.2785121500492096, "sampling/importance_sampling_ratio/mean": 1.0123933553695679, "sampling/importance_sampling_ratio/max": 1.7498754262924194, "entropy": 0.6255333051085472, "clip_ratio/low_mean": 0.0956260934472084, "clip_ratio/low_min": 0.0956260934472084, "clip_ratio/high_mean": 0.02377136843279004, "clip_ratio/high_max": 0.02377136843279004, "clip_ratio/region_mean": 0.11939746187999845, "reward_total_mean": 0.45695754885673523, "reward_meter_mean": 0.6388722658157349, "reward_meter_std": 0.42735129594802856, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9902987480163574, "reward_repeat_soft_std": 0.009319685399532318, "reward_judge_quality_mean": 0.22500000894069672, "reward_judge_quality_std": 0.13887301087379456, "reward_total_composite_mean": 0.45695754885673523, "reward_total_composite_std": 0.11092603206634521} {"timestamp_utc": "2026-04-13T12:21:06Z", "mode": "train", "global_step": 2039, "epoch": 0.2048216976393772, "loss": -0.006, "grad_norm": 15.822030067443848, "learning_rate": 3.8242424242424245e-06, "num_tokens": 3667123.0, "completions/mean_length": 19.5, "completions/min_length": 17.0, "completions/max_length": 22.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 19.5, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 22.0, "rewards/meter/mean": 0.7434380054473877, "rewards/meter/std": 0.25848403573036194, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9545454382896423, "rewards/repeat_soft/std": 0.014728941954672337, "rewards/judge_quality/mean": 0.14625000953674316, "rewards/judge_quality/std": 0.010606604628264904, "rewards/total_composite/mean": 0.413274347782135, "rewards/total_composite/std": 0.022209085524082184, "reward": 0.413274347782135, "reward_std": 0.022209081798791885, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14968574047088623, "sampling/sampling_logp_difference/max": 1.307535171508789, "sampling/importance_sampling_ratio/min": 0.27048593759536743, "sampling/importance_sampling_ratio/mean": 1.002846598625183, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7239391058683395, "clip_ratio/low_mean": 0.07763783447444439, "clip_ratio/low_min": 0.07763783447444439, "clip_ratio/high_mean": 0.05220437468960881, "clip_ratio/high_max": 0.05220437468960881, "clip_ratio/region_mean": 0.1298422091640532, "reward_total_mean": 0.413274347782135, "reward_meter_mean": 0.7434380054473877, "reward_meter_std": 0.25848403573036194, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9545454382896423, "reward_repeat_soft_std": 0.014728941954672337, "reward_judge_quality_mean": 0.14625000953674316, "reward_judge_quality_std": 0.010606604628264904, "reward_total_composite_mean": 0.413274347782135, "reward_total_composite_std": 0.022209085524082184} {"timestamp_utc": "2026-04-13T12:21:13Z", "mode": "train", "global_step": 2040, "epoch": 0.2049221496735309, "loss": 0.0466, "grad_norm": 7.104791164398193, "learning_rate": 3.821212121212122e-06, "num_tokens": 3668966.0, "completions/mean_length": 54.375, "completions/min_length": 46.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.375, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9765020608901978, "rewards/meter/std": 0.016999514773488045, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8627753257751465, "rewards/repeat_soft/std": 0.033737439662218094, "rewards/judge_quality/mean": 0.16500000655651093, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.43422314524650574, "rewards/total_composite/std": 0.011782454326748848, "reward": 0.43422314524650574, "reward_std": 0.011782454326748848, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10320676863193512, "sampling/sampling_logp_difference/max": 1.7726988792419434, "sampling/importance_sampling_ratio/min": 0.169873908162117, "sampling/importance_sampling_ratio/mean": 1.0088846683502197, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5597806870937347, "clip_ratio/low_mean": 0.047498751897364855, "clip_ratio/low_min": 0.047498751897364855, "clip_ratio/high_mean": 0.058153492864221334, "clip_ratio/high_max": 0.058153492864221334, "clip_ratio/region_mean": 0.10565224476158619, "reward_total_mean": 0.43422314524650574, "reward_meter_mean": 0.9765020608901978, "reward_meter_std": 0.016999514773488045, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8627753257751465, "reward_repeat_soft_std": 0.033737439662218094, "reward_judge_quality_mean": 0.16500000655651093, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.43422314524650574, "reward_total_composite_std": 0.011782454326748848} {"timestamp_utc": "2026-04-13T12:21:20Z", "mode": "train", "global_step": 2041, "epoch": 0.20502260170768458, "loss": 0.004, "grad_norm": 8.011926651000977, "learning_rate": 3.818181818181819e-06, "num_tokens": 3671616.0, "completions/mean_length": 138.25, "completions/min_length": 127.0, "completions/max_length": 150.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 138.25, "completions/min_terminated_length": 127.0, "completions/max_terminated_length": 150.0, "rewards/meter/mean": 0.9829698801040649, "rewards/meter/std": 0.004708987195044756, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9096139073371887, "rewards/repeat_soft/std": 0.023690275847911835, "rewards/judge_quality/mean": 0.16875001788139343, "rewards/judge_quality/std": 0.022320719435811043, "rewards/total_composite/mean": 0.4442504644393921, "rewards/total_composite/std": 0.01413500215858221, "reward": 0.4442504644393921, "reward_std": 0.014134998433291912, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0858384296298027, "sampling/sampling_logp_difference/max": 2.0158839225769043, "sampling/importance_sampling_ratio/min": 0.13320261240005493, "sampling/importance_sampling_ratio/mean": 1.0017367601394653, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3669454865157604, "clip_ratio/low_mean": 0.022644356824457645, "clip_ratio/low_min": 0.022644356824457645, "clip_ratio/high_mean": 0.07045797212049365, "clip_ratio/high_max": 0.07045797212049365, "clip_ratio/region_mean": 0.0931023289449513, "reward_total_mean": 0.4442504644393921, "reward_meter_mean": 0.9829698801040649, "reward_meter_std": 0.004708987195044756, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9096139073371887, "reward_repeat_soft_std": 0.023690275847911835, "reward_judge_quality_mean": 0.16875001788139343, "reward_judge_quality_std": 0.022320719435811043, "reward_total_composite_mean": 0.4442504644393921, "reward_total_composite_std": 0.01413500215858221} {"timestamp_utc": "2026-04-13T12:21:27Z", "mode": "train", "global_step": 2042, "epoch": 0.20512305374183828, "loss": 0.0166, "grad_norm": 14.088408470153809, "learning_rate": 3.8151515151515155e-06, "num_tokens": 3673257.0, "completions/mean_length": 36.125, "completions/min_length": 32.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.125, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.8494052290916443, "rewards/meter/std": 0.2127019166946411, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9717245101928711, "rewards/repeat_soft/std": 0.03682416304945946, "rewards/judge_quality/mean": 0.1537500023841858, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.4294499158859253, "rewards/total_composite/std": 0.023655403405427933, "reward": 0.4294499158859253, "reward_std": 0.023655403405427933, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09994067251682281, "sampling/sampling_logp_difference/max": 1.8048105239868164, "sampling/importance_sampling_ratio/min": 0.16450563073158264, "sampling/importance_sampling_ratio/mean": 0.9966678023338318, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3427230603992939, "clip_ratio/low_mean": 0.020891986088827252, "clip_ratio/low_min": 0.020891986088827252, "clip_ratio/high_mean": 0.0672743059694767, "clip_ratio/high_max": 0.0672743059694767, "clip_ratio/region_mean": 0.08816629205830395, "reward_total_mean": 0.4294499158859253, "reward_meter_mean": 0.8494052290916443, "reward_meter_std": 0.2127019166946411, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9717245101928711, "reward_repeat_soft_std": 0.03682416304945946, "reward_judge_quality_mean": 0.1537500023841858, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.4294499158859253, "reward_total_composite_std": 0.023655403405427933} {"timestamp_utc": "2026-04-13T12:21:34Z", "mode": "train", "global_step": 2043, "epoch": 0.20522350577599197, "loss": 0.0335, "grad_norm": 7.588534832000732, "learning_rate": 3.8121212121212127e-06, "num_tokens": 3674964.0, "completions/mean_length": 59.375, "completions/min_length": 54.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.375, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.9905557036399841, "rewards/meter/std": 0.003954746760427952, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9122059345245361, "rewards/repeat_soft/std": 0.05901812016963959, "rewards/judge_quality/mean": 0.1574999988079071, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.4382306933403015, "rewards/total_composite/std": 0.011901028454303741, "reward": 0.4382306933403015, "reward_std": 0.011901023797690868, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07994444668292999, "sampling/sampling_logp_difference/max": 1.2873473167419434, "sampling/importance_sampling_ratio/min": 0.2760019600391388, "sampling/importance_sampling_ratio/mean": 0.9986298084259033, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.41187846660614014, "clip_ratio/low_mean": 0.037985481321811676, "clip_ratio/low_min": 0.037985481321811676, "clip_ratio/high_mean": 0.03362857084721327, "clip_ratio/high_max": 0.03362857084721327, "clip_ratio/region_mean": 0.07161405216902494, "reward_total_mean": 0.4382306933403015, "reward_meter_mean": 0.9905557036399841, "reward_meter_std": 0.003954746760427952, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9122059345245361, "reward_repeat_soft_std": 0.05901812016963959, "reward_judge_quality_mean": 0.1574999988079071, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.4382306933403015, "reward_total_composite_std": 0.011901028454303741} {"timestamp_utc": "2026-04-13T12:21:41Z", "mode": "train", "global_step": 2044, "epoch": 0.20532395781014565, "loss": 0.2662, "grad_norm": 15.66690444946289, "learning_rate": 3.8090909090909095e-06, "num_tokens": 3676748.0, "completions/mean_length": 59.0, "completions/min_length": 45.0, "completions/max_length": 104.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.0, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.6437088847160339, "rewards/meter/std": 0.34713518619537354, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8980014324188232, "rewards/repeat_soft/std": 0.019471298903226852, "rewards/judge_quality/mean": 0.14250001311302185, "rewards/judge_quality/std": 0.013887306675314903, "rewards/total_composite/mean": 0.3829920291900635, "rewards/total_composite/std": 0.0643611028790474, "reward": 0.3829920291900635, "reward_std": 0.0643610954284668, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10684850811958313, "sampling/sampling_logp_difference/max": 2.7498672008514404, "sampling/importance_sampling_ratio/min": 0.06393635272979736, "sampling/importance_sampling_ratio/mean": 1.0027379989624023, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.517706822603941, "clip_ratio/low_mean": 0.04134615417569876, "clip_ratio/low_min": 0.04134615417569876, "clip_ratio/high_mean": 0.06284213112667203, "clip_ratio/high_max": 0.06284213112667203, "clip_ratio/region_mean": 0.10418828530237079, "reward_total_mean": 0.3829920291900635, "reward_meter_mean": 0.6437088847160339, "reward_meter_std": 0.34713518619537354, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8980014324188232, "reward_repeat_soft_std": 0.019471298903226852, "reward_judge_quality_mean": 0.14250001311302185, "reward_judge_quality_std": 0.013887306675314903, "reward_total_composite_mean": 0.3829920291900635, "reward_total_composite_std": 0.0643611028790474} {"timestamp_utc": "2026-04-13T12:21:49Z", "mode": "train", "global_step": 2045, "epoch": 0.20542440984429935, "loss": 0.1668, "grad_norm": 5.523110866546631, "learning_rate": 3.8060606060606064e-06, "num_tokens": 3679184.0, "completions/mean_length": 142.5, "completions/min_length": 102.0, "completions/max_length": 180.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 142.5, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 180.0, "rewards/meter/mean": 0.974332332611084, "rewards/meter/std": 0.021938016638159752, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.842339277267456, "rewards/repeat_soft/std": 0.056603413075208664, "rewards/judge_quality/mean": 0.14250001311302185, "rewards/judge_quality/std": 0.031052954494953156, "rewards/total_composite/mean": 0.37899357080459595, "rewards/total_composite/std": 0.03571334108710289, "reward": 0.37899357080459595, "reward_std": 0.03571334481239319, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08776529133319855, "sampling/sampling_logp_difference/max": 2.9584617614746094, "sampling/importance_sampling_ratio/min": 0.051898688077926636, "sampling/importance_sampling_ratio/mean": 1.0044498443603516, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3657061904668808, "clip_ratio/low_mean": 0.04262035293504596, "clip_ratio/low_min": 0.04262035293504596, "clip_ratio/high_mean": 0.028448880184441805, "clip_ratio/high_max": 0.028448880184441805, "clip_ratio/region_mean": 0.07106923311948776, "reward_total_mean": 0.37899357080459595, "reward_meter_mean": 0.974332332611084, "reward_meter_std": 0.021938016638159752, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.842339277267456, "reward_repeat_soft_std": 0.056603413075208664, "reward_judge_quality_mean": 0.14250001311302185, "reward_judge_quality_std": 0.031052954494953156, "reward_total_composite_mean": 0.37899357080459595, "reward_total_composite_std": 0.03571334108710289} {"timestamp_utc": "2026-04-13T12:21:56Z", "mode": "train", "global_step": 2046, "epoch": 0.20552486187845304, "loss": 0.1164, "grad_norm": 6.8209147453308105, "learning_rate": 3.803030303030303e-06, "num_tokens": 3681523.0, "completions/mean_length": 115.375, "completions/min_length": 101.0, "completions/max_length": 147.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.375, "completions/min_terminated_length": 101.0, "completions/max_terminated_length": 147.0, "rewards/meter/mean": 0.9768247604370117, "rewards/meter/std": 0.0234832800924778, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8576970100402832, "rewards/repeat_soft/std": 0.06382130086421967, "rewards/judge_quality/mean": 0.14250001311302185, "rewards/judge_quality/std": 0.02121320739388466, "rewards/total_composite/mean": 0.40669384598731995, "rewards/total_composite/std": 0.035279784351587296, "reward": 0.40669384598731995, "reward_std": 0.035279788076877594, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10315954685211182, "sampling/sampling_logp_difference/max": 4.479145526885986, "sampling/importance_sampling_ratio/min": 0.011343101970851421, "sampling/importance_sampling_ratio/mean": 0.9935485124588013, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4440900795161724, "clip_ratio/low_mean": 0.0415108953602612, "clip_ratio/low_min": 0.0415108953602612, "clip_ratio/high_mean": 0.05447726044803858, "clip_ratio/high_max": 0.05447726044803858, "clip_ratio/region_mean": 0.09598815580829978, "reward_total_mean": 0.40669384598731995, "reward_meter_mean": 0.9768247604370117, "reward_meter_std": 0.0234832800924778, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8576970100402832, "reward_repeat_soft_std": 0.06382130086421967, "reward_judge_quality_mean": 0.14250001311302185, "reward_judge_quality_std": 0.02121320739388466, "reward_total_composite_mean": 0.40669384598731995, "reward_total_composite_std": 0.035279784351587296} {"timestamp_utc": "2026-04-13T12:22:07Z", "mode": "train", "global_step": 2047, "epoch": 0.20562531391260672, "loss": -0.1227, "grad_norm": 1.9844030141830444, "learning_rate": 3.8000000000000005e-06, "num_tokens": 3683165.0, "completions/mean_length": 100.25, "completions/min_length": 36.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 41.42857360839844, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.8852245807647705, "rewards/meter/std": 0.30692076683044434, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8760888576507568, "rewards/repeat_soft/std": 0.06388038396835327, "rewards/judge_quality/mean": 0.13750000298023224, "rewards/judge_quality/std": 0.0353553406894207, "rewards/total_composite/mean": 0.41670188307762146, "rewards/total_composite/std": 0.028256313875317574, "reward": 0.41670188307762146, "reward_std": 0.02825631946325302, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07085798680782318, "sampling/sampling_logp_difference/max": 0.6945812702178955, "sampling/importance_sampling_ratio/min": 0.5122395157814026, "sampling/importance_sampling_ratio/mean": 1.0203478336334229, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.38697320967912674, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.05263052717782557, "clip_ratio/high_max": 0.05263052717782557, "clip_ratio/region_mean": 0.05263052717782557, "reward_total_mean": 0.41670188307762146, "reward_meter_mean": 0.8852245807647705, "reward_meter_std": 0.30692076683044434, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8760888576507568, "reward_repeat_soft_std": 0.06388038396835327, "reward_judge_quality_mean": 0.13750000298023224, "reward_judge_quality_std": 0.0353553406894207, "reward_total_composite_mean": 0.41670188307762146, "reward_total_composite_std": 0.028256313875317574} {"timestamp_utc": "2026-04-13T12:22:14Z", "mode": "train", "global_step": 2048, "epoch": 0.20572576594676042, "loss": -0.0277, "grad_norm": 7.260988235473633, "learning_rate": 3.7969696969696973e-06, "num_tokens": 3685202.0, "completions/mean_length": 74.625, "completions/min_length": 64.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.625, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.9884733557701111, "rewards/meter/std": 0.009229067713022232, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8281691670417786, "rewards/repeat_soft/std": 0.028537092730402946, "rewards/judge_quality/mean": 0.21000000834465027, "rewards/judge_quality/std": 0.08485280722379684, "rewards/total_composite/mean": 0.45912399888038635, "rewards/total_composite/std": 0.05568560212850571, "reward": 0.45912399888038635, "reward_std": 0.05568559467792511, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07889800518751144, "sampling/sampling_logp_difference/max": 1.560380458831787, "sampling/importance_sampling_ratio/min": 0.2100561261177063, "sampling/importance_sampling_ratio/mean": 1.0017101764678955, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.46210258826613426, "clip_ratio/low_mean": 0.06502922880463302, "clip_ratio/low_min": 0.06502922880463302, "clip_ratio/high_mean": 0.0074404762126505375, "clip_ratio/high_max": 0.0074404762126505375, "clip_ratio/region_mean": 0.07246970501728356, "reward_total_mean": 0.45912399888038635, "reward_meter_mean": 0.9884733557701111, "reward_meter_std": 0.009229067713022232, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8281691670417786, "reward_repeat_soft_std": 0.028537092730402946, "reward_judge_quality_mean": 0.21000000834465027, "reward_judge_quality_std": 0.08485280722379684, "reward_total_composite_mean": 0.45912399888038635, "reward_total_composite_std": 0.05568560212850571} {"timestamp_utc": "2026-04-13T12:22:20Z", "mode": "train", "global_step": 2049, "epoch": 0.2058262179809141, "loss": -0.0284, "grad_norm": 10.234697341918945, "learning_rate": 3.793939393939394e-06, "num_tokens": 3687043.0, "completions/mean_length": 63.125, "completions/min_length": 54.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.125, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.2560454308986664, "rewards/meter/std": 0.2626054286956787, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8586742281913757, "rewards/repeat_soft/std": 0.03004816733300686, "rewards/judge_quality/mean": 0.16124999523162842, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.35626357793807983, "rewards/total_composite/std": 0.027345743030309677, "reward": 0.35626357793807983, "reward_std": 0.027345741167664528, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0907815545797348, "sampling/sampling_logp_difference/max": 1.9165549278259277, "sampling/importance_sampling_ratio/min": 0.1471129208803177, "sampling/importance_sampling_ratio/mean": 0.9916093945503235, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.33177791349589825, "clip_ratio/low_mean": 0.04983414337038994, "clip_ratio/low_min": 0.04983414337038994, "clip_ratio/high_mean": 0.028859290294349194, "clip_ratio/high_max": 0.028859290294349194, "clip_ratio/region_mean": 0.07869343366473913, "reward_total_mean": 0.35626357793807983, "reward_meter_mean": 0.2560454308986664, "reward_meter_std": 0.2626054286956787, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8586742281913757, "reward_repeat_soft_std": 0.03004816733300686, "reward_judge_quality_mean": 0.16124999523162842, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.35626357793807983, "reward_total_composite_std": 0.027345743030309677} {"timestamp_utc": "2026-04-13T12:22:28Z", "mode": "train", "global_step": 2050, "epoch": 0.20592667001506781, "loss": 0.1111, "grad_norm": 7.519757270812988, "learning_rate": 3.7909090909090914e-06, "num_tokens": 3689510.0, "completions/mean_length": 124.375, "completions/min_length": 104.0, "completions/max_length": 154.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 124.375, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 154.0, "rewards/meter/mean": 0.7648420333862305, "rewards/meter/std": 0.34862419962882996, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8720245957374573, "rewards/repeat_soft/std": 0.041297126561403275, "rewards/judge_quality/mean": 0.1574999988079071, "rewards/judge_quality/std": 0.031052954494953156, "rewards/total_composite/mean": 0.39393067359924316, "rewards/total_composite/std": 0.030269678682088852, "reward": 0.39393067359924316, "reward_std": 0.030269671231508255, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11565042287111282, "sampling/sampling_logp_difference/max": 2.13187837600708, "sampling/importance_sampling_ratio/min": 0.11861427873373032, "sampling/importance_sampling_ratio/mean": 1.0017586946487427, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5740426108241081, "clip_ratio/low_mean": 0.02805569628253579, "clip_ratio/low_min": 0.02805569628253579, "clip_ratio/high_mean": 0.062328214291483164, "clip_ratio/high_max": 0.062328214291483164, "clip_ratio/region_mean": 0.09038391057401896, "reward_total_mean": 0.39393067359924316, "reward_meter_mean": 0.7648420333862305, "reward_meter_std": 0.34862419962882996, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8720245957374573, "reward_repeat_soft_std": 0.041297126561403275, "reward_judge_quality_mean": 0.1574999988079071, "reward_judge_quality_std": 0.031052954494953156, "reward_total_composite_mean": 0.39393067359924316, "reward_total_composite_std": 0.030269678682088852} {"timestamp_utc": "2026-04-13T12:23:30Z", "mode": "eval", "global_step": 2050, "epoch": 0.20592667001506781, "eval_loss": NaN, "eval_runtime": 61.6625, "eval_samples_per_second": 1.297, "eval_steps_per_second": 0.162, "eval_num_tokens": 3689510.0, "eval_completions/mean_length": 104.15, "eval_completions/min_length": 32.8, "eval_completions/max_length": 299.5, "eval_completions/clipped_ratio": 0.05, "eval_completions/mean_terminated_length": 82.56428756713868, "eval_completions/min_terminated_length": 32.8, "eval_completions/max_terminated_length": 152.1, "eval_rewards/meter/mean": 0.8470357298851013, "eval_rewards/meter/std": 0.210162253677845, "eval_rewards/count_adherence/mean": 0.9524999976158142, "eval_rewards/count_adherence/std": 0.10983712449669839, "eval_rewards/hard_gate/mean": 0.95, "eval_rewards/hard_gate/std": 0.1414213538169861, "eval_rewards/repeat_soft/mean": 0.881467753648758, "eval_rewards/repeat_soft/std": 0.07334110662341117, "eval_rewards/judge_quality/mean": 0.15550000816583634, "eval_rewards/judge_quality/std": 0.028039283119142056, "eval_rewards/total_composite/mean": 0.39409635961055756, "eval_rewards/total_composite/std": 0.07721717413514853, "eval_reward": 0.39409635961055756, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.0478603333234787, "eval_sampling/sampling_logp_difference/max": 1.110507845878601, "eval_sampling/importance_sampling_ratio/min": 0.35095261931419375, "eval_sampling/importance_sampling_ratio/mean": 1.0076323390007018, "eval_sampling/importance_sampling_ratio/max": 1.4491962432861327, "eval_entropy": 0.48588710129261015, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.39409635961055756, "eval_reward_meter_mean": 0.8470357298851013, "eval_reward_meter_std": 0.210162253677845, "eval_reward_count_adherence_mean": 0.9524999976158142, "eval_reward_count_adherence_std": 0.10983712449669839, "eval_reward_hard_gate_mean": 0.95, "eval_reward_hard_gate_std": 0.1414213538169861, "eval_reward_repeat_soft_mean": 0.881467753648758, "eval_reward_repeat_soft_std": 0.07334110662341117, "eval_reward_judge_quality_mean": 0.15550000816583634, "eval_reward_judge_quality_std": 0.028039283119142056, "eval_reward_total_composite_mean": 0.39409635961055756, "eval_reward_total_composite_std": 0.07721717413514853} {"timestamp_utc": "2026-04-13T12:23:39Z", "mode": "train", "global_step": 2051, "epoch": 0.2060271220492215, "loss": 0.0993, "grad_norm": 9.6243257522583, "learning_rate": 3.7878787878787882e-06, "num_tokens": 3691213.0, "completions/mean_length": 55.875, "completions/min_length": 48.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.875, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9860696792602539, "rewards/meter/std": 0.010559221729636192, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9134825468063354, "rewards/repeat_soft/std": 0.0551791712641716, "rewards/judge_quality/mean": 0.23250001668930054, "rewards/judge_quality/std": 0.1348808854818344, "rewards/total_composite/mean": 0.4856888949871063, "rewards/total_composite/std": 0.09246828407049179, "reward": 0.4856888949871063, "reward_std": 0.09246828407049179, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10899222642183304, "sampling/sampling_logp_difference/max": 1.9581661224365234, "sampling/importance_sampling_ratio/min": 0.14111697673797607, "sampling/importance_sampling_ratio/mean": 1.0144563913345337, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5422843880951405, "clip_ratio/low_mean": 0.07558443117886782, "clip_ratio/low_min": 0.07558443117886782, "clip_ratio/high_mean": 0.01791666680946946, "clip_ratio/high_max": 0.01791666680946946, "clip_ratio/region_mean": 0.09350109798833728, "reward_total_mean": 0.4856888949871063, "reward_meter_mean": 0.9860696792602539, "reward_meter_std": 0.010559221729636192, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9134825468063354, "reward_repeat_soft_std": 0.0551791712641716, "reward_judge_quality_mean": 0.23250001668930054, "reward_judge_quality_std": 0.1348808854818344, "reward_total_composite_mean": 0.4856888949871063, "reward_total_composite_std": 0.09246828407049179} {"timestamp_utc": "2026-04-13T12:23:51Z", "mode": "train", "global_step": 2052, "epoch": 0.20612757408337518, "loss": -0.0861, "grad_norm": 3.488429307937622, "learning_rate": 3.784848484848485e-06, "num_tokens": 3693351.0, "completions/mean_length": 158.25, "completions/min_length": 85.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 107.71428680419922, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 132.0, "rewards/meter/mean": 0.7394720911979675, "rewards/meter/std": 0.34064599871635437, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.8292969465255737, "rewards/repeat_soft/std": 0.09894319623708725, "rewards/judge_quality/mean": 0.14125001430511475, "rewards/judge_quality/std": 0.04454132169485092, "rewards/total_composite/mean": 0.294907808303833, "rewards/total_composite/std": 0.18495747447013855, "reward": 0.294907808303833, "reward_std": 0.18495745956897736, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08565527200698853, "sampling/sampling_logp_difference/max": 2.172407865524292, "sampling/importance_sampling_ratio/min": 0.11390302330255508, "sampling/importance_sampling_ratio/mean": 1.0037480592727661, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4531273767352104, "clip_ratio/low_mean": 0.007575757801532745, "clip_ratio/low_min": 0.007575757801532745, "clip_ratio/high_mean": 0.06291623646393418, "clip_ratio/high_max": 0.06291623646393418, "clip_ratio/region_mean": 0.07049199426546693, "reward_total_mean": 0.294907808303833, "reward_meter_mean": 0.7394720911979675, "reward_meter_std": 0.34064599871635437, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.8292969465255737, "reward_repeat_soft_std": 0.09894319623708725, "reward_judge_quality_mean": 0.14125001430511475, "reward_judge_quality_std": 0.04454132169485092, "reward_total_composite_mean": 0.294907808303833, "reward_total_composite_std": 0.18495747447013855} {"timestamp_utc": "2026-04-13T12:23:59Z", "mode": "train", "global_step": 2053, "epoch": 0.20622802611752888, "loss": 0.0299, "grad_norm": 19.44232177734375, "learning_rate": 3.781818181818182e-06, "num_tokens": 3694749.0, "completions/mean_length": 18.75, "completions/min_length": 17.0, "completions/max_length": 25.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 18.75, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 25.0, "rewards/meter/mean": 0.9142047166824341, "rewards/meter/std": 0.13621951639652252, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9537500143051147, "rewards/repeat_soft/std": 0.02474873699247837, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4321974515914917, "rewards/total_composite/std": 0.013160732574760914, "reward": 0.4321974515914917, "reward_std": 0.013160724192857742, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06499005854129791, "sampling/sampling_logp_difference/max": 1.1320464611053467, "sampling/importance_sampling_ratio/min": 0.3223728537559509, "sampling/importance_sampling_ratio/mean": 1.0129449367523193, "sampling/importance_sampling_ratio/max": 1.500510811805725, "entropy": 0.4856715239584446, "clip_ratio/low_mean": 0.047412280924618244, "clip_ratio/low_min": 0.047412280924618244, "clip_ratio/high_mean": 0.03345588315278292, "clip_ratio/high_max": 0.03345588315278292, "clip_ratio/region_mean": 0.08086816407740116, "reward_total_mean": 0.4321974515914917, "reward_meter_mean": 0.9142047166824341, "reward_meter_std": 0.13621951639652252, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9537500143051147, "reward_repeat_soft_std": 0.02474873699247837, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4321974515914917, "reward_total_composite_std": 0.013160732574760914} {"timestamp_utc": "2026-04-13T12:24:05Z", "mode": "train", "global_step": 2054, "epoch": 0.20632847815168257, "loss": 0.0103, "grad_norm": 7.740578651428223, "learning_rate": 3.778787878787879e-06, "num_tokens": 3696438.0, "completions/mean_length": 63.125, "completions/min_length": 54.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.125, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9650364518165588, "rewards/meter/std": 0.018497325479984283, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9724728465080261, "rewards/repeat_soft/std": 0.017807554453611374, "rewards/judge_quality/mean": 0.16625000536441803, "rewards/judge_quality/std": 0.035431019961833954, "rewards/total_composite/mean": 0.4502558708190918, "rewards/total_composite/std": 0.023599883541464806, "reward": 0.4502558708190918, "reward_std": 0.023599879816174507, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0975913405418396, "sampling/sampling_logp_difference/max": 1.569288969039917, "sampling/importance_sampling_ratio/min": 0.20819316804409027, "sampling/importance_sampling_ratio/mean": 1.0246812105178833, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4856261573731899, "clip_ratio/low_mean": 0.054329875856637955, "clip_ratio/low_min": 0.054329875856637955, "clip_ratio/high_mean": 0.0265895277261734, "clip_ratio/high_max": 0.0265895277261734, "clip_ratio/region_mean": 0.08091940358281136, "reward_total_mean": 0.4502558708190918, "reward_meter_mean": 0.9650364518165588, "reward_meter_std": 0.018497325479984283, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9724728465080261, "reward_repeat_soft_std": 0.017807554453611374, "reward_judge_quality_mean": 0.16625000536441803, "reward_judge_quality_std": 0.035431019961833954, "reward_total_composite_mean": 0.4502558708190918, "reward_total_composite_std": 0.023599883541464806} {"timestamp_utc": "2026-04-13T12:24:13Z", "mode": "train", "global_step": 2055, "epoch": 0.20642893018583627, "loss": 0.1448, "grad_norm": 7.970420837402344, "learning_rate": 3.775757575757576e-06, "num_tokens": 3698532.0, "completions/mean_length": 88.75, "completions/min_length": 72.0, "completions/max_length": 122.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 88.75, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 122.0, "rewards/meter/mean": 0.9468092322349548, "rewards/meter/std": 0.06834986060857773, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9011238217353821, "rewards/repeat_soft/std": 0.041544847190380096, "rewards/judge_quality/mean": 0.18000000715255737, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.43969523906707764, "rewards/total_composite/std": 0.019526204094290733, "reward": 0.43969523906707764, "reward_std": 0.019526205956935883, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09225019812583923, "sampling/sampling_logp_difference/max": 2.155517101287842, "sampling/importance_sampling_ratio/min": 0.1158432736992836, "sampling/importance_sampling_ratio/mean": 1.0105295181274414, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4102574083954096, "clip_ratio/low_mean": 0.030017105396836996, "clip_ratio/low_min": 0.030017105396836996, "clip_ratio/high_mean": 0.04154863650910556, "clip_ratio/high_max": 0.04154863650910556, "clip_ratio/region_mean": 0.07156574190594256, "reward_total_mean": 0.43969523906707764, "reward_meter_mean": 0.9468092322349548, "reward_meter_std": 0.06834986060857773, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9011238217353821, "reward_repeat_soft_std": 0.041544847190380096, "reward_judge_quality_mean": 0.18000000715255737, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.43969523906707764, "reward_total_composite_std": 0.019526204094290733} {"timestamp_utc": "2026-04-13T12:24:25Z", "mode": "train", "global_step": 2056, "epoch": 0.20652938221998995, "loss": -0.2314, "grad_norm": 1.922808051109314, "learning_rate": 3.772727272727273e-06, "num_tokens": 3701297.0, "completions/mean_length": 198.625, "completions/min_length": 137.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 153.85714721679688, "completions/min_terminated_length": 137.0, "completions/max_terminated_length": 167.0, "rewards/meter/mean": 0.8928427696228027, "rewards/meter/std": 0.1732001155614853, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.1414213478565216, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9112764596939087, "rewards/repeat_soft/std": 0.03366226330399513, "rewards/judge_quality/mean": 0.15625, "rewards/judge_quality/std": 0.047790467739105225, "rewards/total_composite/mean": 0.37900832295417786, "rewards/total_composite/std": 0.1544741690158844, "reward": 0.37900832295417786, "reward_std": 0.1544741690158844, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11159887164831161, "sampling/sampling_logp_difference/max": 2.3941075801849365, "sampling/importance_sampling_ratio/min": 0.1788671761751175, "sampling/importance_sampling_ratio/mean": 1.0042688846588135, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.49233855307102203, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.10130473598837852, "clip_ratio/high_max": 0.10130473598837852, "clip_ratio/region_mean": 0.10130473598837852, "reward_total_mean": 0.37900832295417786, "reward_meter_mean": 0.8928427696228027, "reward_meter_std": 0.1732001155614853, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.1414213478565216, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9112764596939087, "reward_repeat_soft_std": 0.03366226330399513, "reward_judge_quality_mean": 0.15625, "reward_judge_quality_std": 0.047790467739105225, "reward_total_composite_mean": 0.37900832295417786, "reward_total_composite_std": 0.1544741690158844} {"timestamp_utc": "2026-04-13T12:24:31Z", "mode": "train", "global_step": 2057, "epoch": 0.20662983425414364, "loss": 0.0255, "grad_norm": 11.074159622192383, "learning_rate": 3.76969696969697e-06, "num_tokens": 3702852.0, "completions/mean_length": 41.375, "completions/min_length": 32.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.375, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9662160277366638, "rewards/meter/std": 0.046560220420360565, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9596341848373413, "rewards/repeat_soft/std": 0.04224960133433342, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.43815118074417114, "rewards/total_composite/std": 0.007245528046041727, "reward": 0.43815118074417114, "reward_std": 0.00724552758038044, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09605085104703903, "sampling/sampling_logp_difference/max": 1.3370459079742432, "sampling/importance_sampling_ratio/min": 0.26262032985687256, "sampling/importance_sampling_ratio/mean": 1.0031450986862183, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.575896430760622, "clip_ratio/low_mean": 0.02825218066573143, "clip_ratio/low_min": 0.02825218066573143, "clip_ratio/high_mean": 0.08676579128950834, "clip_ratio/high_max": 0.08676579128950834, "clip_ratio/region_mean": 0.11501797195523977, "reward_total_mean": 0.43815118074417114, "reward_meter_mean": 0.9662160277366638, "reward_meter_std": 0.046560220420360565, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9596341848373413, "reward_repeat_soft_std": 0.04224960133433342, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.43815118074417114, "reward_total_composite_std": 0.007245528046041727} {"timestamp_utc": "2026-04-13T12:24:38Z", "mode": "train", "global_step": 2058, "epoch": 0.20673028628829734, "loss": 0.0517, "grad_norm": 14.113903045654297, "learning_rate": 3.766666666666667e-06, "num_tokens": 3704402.0, "completions/mean_length": 26.75, "completions/min_length": 23.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.75, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.6191296577453613, "rewards/meter/std": 0.3962414264678955, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9412171840667725, "rewards/repeat_soft/std": 0.030817119404673576, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.40154772996902466, "rewards/total_composite/std": 0.037522632628679276, "reward": 0.40154772996902466, "reward_std": 0.03752264380455017, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11040031164884567, "sampling/sampling_logp_difference/max": 1.6234591007232666, "sampling/importance_sampling_ratio/min": 0.2405642867088318, "sampling/importance_sampling_ratio/mean": 1.016933560371399, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6415188424289227, "clip_ratio/low_mean": 0.050534188747406006, "clip_ratio/low_min": 0.050534188747406006, "clip_ratio/high_mean": 0.06094925617799163, "clip_ratio/high_max": 0.06094925617799163, "clip_ratio/region_mean": 0.11148344492539763, "reward_total_mean": 0.40154772996902466, "reward_meter_mean": 0.6191296577453613, "reward_meter_std": 0.3962414264678955, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9412171840667725, "reward_repeat_soft_std": 0.030817119404673576, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.40154772996902466, "reward_total_composite_std": 0.037522632628679276} {"timestamp_utc": "2026-04-13T12:24:49Z", "mode": "train", "global_step": 2059, "epoch": 0.20683073832245102, "loss": -0.129, "grad_norm": 1.6379417181015015, "learning_rate": 3.7636363636363637e-06, "num_tokens": 3706168.0, "completions/mean_length": 101.75, "completions/min_length": 38.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 43.142860412597656, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.8695842027664185, "rewards/meter/std": 0.35068050026893616, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8757412433624268, "rewards/repeat_soft/std": 0.07615207135677338, "rewards/judge_quality/mean": 0.13375000655651093, "rewards/judge_quality/std": 0.03543102368712425, "rewards/total_composite/mean": 0.3706586956977844, "rewards/total_composite/std": 0.15045024454593658, "reward": 0.3706586956977844, "reward_std": 0.1504502296447754, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09259120374917984, "sampling/sampling_logp_difference/max": 1.4251604080200195, "sampling/importance_sampling_ratio/min": 0.24046988785266876, "sampling/importance_sampling_ratio/mean": 0.992057740688324, "sampling/importance_sampling_ratio/max": 1.8282526731491089, "entropy": 0.35852643474936485, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09153049066662788, "clip_ratio/high_max": 0.09153049066662788, "clip_ratio/region_mean": 0.09153049066662788, "reward_total_mean": 0.3706586956977844, "reward_meter_mean": 0.8695842027664185, "reward_meter_std": 0.35068050026893616, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8757412433624268, "reward_repeat_soft_std": 0.07615207135677338, "reward_judge_quality_mean": 0.13375000655651093, "reward_judge_quality_std": 0.03543102368712425, "reward_total_composite_mean": 0.3706586956977844, "reward_total_composite_std": 0.15045024454593658} {"timestamp_utc": "2026-04-13T12:24:55Z", "mode": "train", "global_step": 2060, "epoch": 0.20693119035660473, "loss": 0.103, "grad_norm": 12.3268461227417, "learning_rate": 3.7606060606060605e-06, "num_tokens": 3707710.0, "completions/mean_length": 31.75, "completions/min_length": 24.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.75, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.5625284910202026, "rewards/meter/std": 0.37623330950737, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9688060283660889, "rewards/repeat_soft/std": 0.020261460915207863, "rewards/judge_quality/mean": 0.3425000309944153, "rewards/judge_quality/std": 0.3564407527446747, "rewards/total_composite/mean": 0.47080230712890625, "rewards/total_composite/std": 0.1848701536655426, "reward": 0.47080230712890625, "reward_std": 0.1848701536655426, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11296968907117844, "sampling/sampling_logp_difference/max": 2.78413724899292, "sampling/importance_sampling_ratio/min": 0.061782367527484894, "sampling/importance_sampling_ratio/mean": 1.0109044313430786, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4610154442489147, "clip_ratio/low_mean": 0.08212722046300769, "clip_ratio/low_min": 0.08212722046300769, "clip_ratio/high_mean": 0.02083333395421505, "clip_ratio/high_max": 0.02083333395421505, "clip_ratio/region_mean": 0.10296055441722274, "reward_total_mean": 0.47080230712890625, "reward_meter_mean": 0.5625284910202026, "reward_meter_std": 0.37623330950737, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9688060283660889, "reward_repeat_soft_std": 0.020261460915207863, "reward_judge_quality_mean": 0.3425000309944153, "reward_judge_quality_std": 0.3564407527446747, "reward_total_composite_mean": 0.47080230712890625, "reward_total_composite_std": 0.1848701536655426} {"timestamp_utc": "2026-04-13T12:25:02Z", "mode": "train", "global_step": 2061, "epoch": 0.2070316423907584, "loss": -0.0142, "grad_norm": 10.08353328704834, "learning_rate": 3.757575757575758e-06, "num_tokens": 3709564.0, "completions/mean_length": 63.75, "completions/min_length": 55.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.75, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.47723761200904846, "rewards/meter/std": 0.39061424136161804, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9800647497177124, "rewards/repeat_soft/std": 0.0135366041213274, "rewards/judge_quality/mean": 0.1912500113248825, "rewards/judge_quality/std": 0.10507649928331375, "rewards/total_composite/mean": 0.40029656887054443, "rewards/total_composite/std": 0.038408465683460236, "reward": 0.40029656887054443, "reward_std": 0.03840846195816994, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12142689526081085, "sampling/sampling_logp_difference/max": 1.6213834285736084, "sampling/importance_sampling_ratio/min": 0.19762511551380157, "sampling/importance_sampling_ratio/mean": 0.9930986762046814, "sampling/importance_sampling_ratio/max": 1.9706506729125977, "entropy": 0.8012176007032394, "clip_ratio/low_mean": 0.06417649798095226, "clip_ratio/low_min": 0.06417649798095226, "clip_ratio/high_mean": 0.05263974145054817, "clip_ratio/high_max": 0.05263974145054817, "clip_ratio/region_mean": 0.11681623943150043, "reward_total_mean": 0.40029656887054443, "reward_meter_mean": 0.47723761200904846, "reward_meter_std": 0.39061424136161804, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9800647497177124, "reward_repeat_soft_std": 0.0135366041213274, "reward_judge_quality_mean": 0.1912500113248825, "reward_judge_quality_std": 0.10507649928331375, "reward_total_composite_mean": 0.40029656887054443, "reward_total_composite_std": 0.038408465683460236} {"timestamp_utc": "2026-04-13T12:25:08Z", "mode": "train", "global_step": 2062, "epoch": 0.2071320944249121, "loss": 0.0719, "grad_norm": 10.390212059020996, "learning_rate": 3.7545454545454546e-06, "num_tokens": 3711010.0, "completions/mean_length": 28.75, "completions/min_length": 24.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.75, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.652949869632721, "rewards/meter/std": 0.3838459551334381, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9330303072929382, "rewards/repeat_soft/std": 0.027116809040308, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.40361714363098145, "rewards/total_composite/std": 0.037867337465286255, "reward": 0.40361714363098145, "reward_std": 0.03786734119057655, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06865010410547256, "sampling/sampling_logp_difference/max": 0.7846975326538086, "sampling/importance_sampling_ratio/min": 0.456257700920105, "sampling/importance_sampling_ratio/mean": 1.0090291500091553, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3845411613583565, "clip_ratio/low_mean": 0.034642857033759356, "clip_ratio/low_min": 0.034642857033759356, "clip_ratio/high_mean": 0.03498500818386674, "clip_ratio/high_max": 0.03498500818386674, "clip_ratio/region_mean": 0.0696278652176261, "reward_total_mean": 0.40361714363098145, "reward_meter_mean": 0.652949869632721, "reward_meter_std": 0.3838459551334381, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9330303072929382, "reward_repeat_soft_std": 0.027116809040308, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.40361714363098145, "reward_total_composite_std": 0.037867337465286255} {"timestamp_utc": "2026-04-13T12:25:21Z", "mode": "train", "global_step": 2063, "epoch": 0.2072325464590658, "loss": -0.2095, "grad_norm": 1.7805317640304565, "learning_rate": 3.7515151515151515e-06, "num_tokens": 3713276.0, "completions/mean_length": 175.25, "completions/min_length": 113.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 127.14286041259766, "completions/min_terminated_length": 113.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.761021614074707, "rewards/meter/std": 0.19159801304340363, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.95693039894104, "rewards/repeat_soft/std": 0.021961698308587074, "rewards/judge_quality/mean": 0.15625, "rewards/judge_quality/std": 0.047790467739105225, "rewards/total_composite/mean": 0.3775297999382019, "rewards/total_composite/std": 0.15459077060222626, "reward": 0.3775297999382019, "reward_std": 0.15459077060222626, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11826728284358978, "sampling/sampling_logp_difference/max": 2.1097679138183594, "sampling/importance_sampling_ratio/min": 0.12126611173152924, "sampling/importance_sampling_ratio/mean": 1.017488718032837, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6875141113996506, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09829550376161933, "clip_ratio/high_max": 0.09829550376161933, "clip_ratio/region_mean": 0.09829550376161933, "reward_total_mean": 0.3775297999382019, "reward_meter_mean": 0.761021614074707, "reward_meter_std": 0.19159801304340363, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.95693039894104, "reward_repeat_soft_std": 0.021961698308587074, "reward_judge_quality_mean": 0.15625, "reward_judge_quality_std": 0.047790467739105225, "reward_total_composite_mean": 0.3775297999382019, "reward_total_composite_std": 0.15459077060222626} {"timestamp_utc": "2026-04-13T12:25:28Z", "mode": "train", "global_step": 2064, "epoch": 0.20733299849321948, "loss": 0.1502, "grad_norm": 19.078561782836914, "learning_rate": 3.748484848484849e-06, "num_tokens": 3714667.0, "completions/mean_length": 18.875, "completions/min_length": 14.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 18.875, "completions/min_terminated_length": 14.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.5398879051208496, "rewards/meter/std": 0.44889894127845764, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9491124749183655, "rewards/repeat_soft/std": 0.02061731554567814, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.3950059413909912, "rewards/total_composite/std": 0.043626610189676285, "reward": 0.3950059413909912, "reward_std": 0.04362659528851509, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10711126774549484, "sampling/sampling_logp_difference/max": 0.8745580911636353, "sampling/importance_sampling_ratio/min": 0.41704627871513367, "sampling/importance_sampling_ratio/mean": 1.0046395063400269, "sampling/importance_sampling_ratio/max": 1.696731448173523, "entropy": 0.8246192522346973, "clip_ratio/low_mean": 0.038082108832895756, "clip_ratio/low_min": 0.038082108832895756, "clip_ratio/high_mean": 0.039938801899552345, "clip_ratio/high_max": 0.039938801899552345, "clip_ratio/region_mean": 0.0780209107324481, "reward_total_mean": 0.3950059413909912, "reward_meter_mean": 0.5398879051208496, "reward_meter_std": 0.44889894127845764, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9491124749183655, "reward_repeat_soft_std": 0.02061731554567814, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.3950059413909912, "reward_total_composite_std": 0.043626610189676285} {"timestamp_utc": "2026-04-13T12:25:35Z", "mode": "train", "global_step": 2065, "epoch": 0.2074334505273732, "loss": 0.066, "grad_norm": 13.049152374267578, "learning_rate": 3.745454545454546e-06, "num_tokens": 3716331.0, "completions/mean_length": 40.0, "completions/min_length": 30.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.0, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.4842578172683716, "rewards/meter/std": 0.4082338809967041, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9329119920730591, "rewards/repeat_soft/std": 0.08886377513408661, "rewards/judge_quality/mean": 0.1875, "rewards/judge_quality/std": 0.1060660108923912, "rewards/total_composite/mean": 0.38721537590026855, "rewards/total_composite/std": 0.033759042620658875, "reward": 0.38721537590026855, "reward_std": 0.03375904634594917, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10694070905447006, "sampling/sampling_logp_difference/max": 1.6021480560302734, "sampling/importance_sampling_ratio/min": 0.20146329700946808, "sampling/importance_sampling_ratio/mean": 1.0106390714645386, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.49556347355246544, "clip_ratio/low_mean": 0.0787456389516592, "clip_ratio/low_min": 0.0787456389516592, "clip_ratio/high_mean": 0.03409310942515731, "clip_ratio/high_max": 0.03409310942515731, "clip_ratio/region_mean": 0.11283874837681651, "reward_total_mean": 0.38721537590026855, "reward_meter_mean": 0.4842578172683716, "reward_meter_std": 0.4082338809967041, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9329119920730591, "reward_repeat_soft_std": 0.08886377513408661, "reward_judge_quality_mean": 0.1875, "reward_judge_quality_std": 0.1060660108923912, "reward_total_composite_mean": 0.38721537590026855, "reward_total_composite_std": 0.033759042620658875} {"timestamp_utc": "2026-04-13T12:25:41Z", "mode": "train", "global_step": 2066, "epoch": 0.20753390256152687, "loss": 0.1272, "grad_norm": 16.58196449279785, "learning_rate": 3.742424242424243e-06, "num_tokens": 3717834.0, "completions/mean_length": 21.875, "completions/min_length": 18.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.875, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.9853771328926086, "rewards/meter/std": 0.013587907887995243, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.922544002532959, "rewards/repeat_soft/std": 0.047796379774808884, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.43445590138435364, "rewards/total_composite/std": 0.006304551847279072, "reward": 0.43445590138435364, "reward_std": 0.006304553709924221, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08291006088256836, "sampling/sampling_logp_difference/max": 1.1904220581054688, "sampling/importance_sampling_ratio/min": 0.36182835698127747, "sampling/importance_sampling_ratio/mean": 1.013149619102478, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.48387037962675095, "clip_ratio/low_mean": 0.013829023111611605, "clip_ratio/low_min": 0.013829023111611605, "clip_ratio/high_mean": 0.03735047811642289, "clip_ratio/high_max": 0.03735047811642289, "clip_ratio/region_mean": 0.051179501228034496, "reward_total_mean": 0.43445590138435364, "reward_meter_mean": 0.9853771328926086, "reward_meter_std": 0.013587907887995243, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.922544002532959, "reward_repeat_soft_std": 0.047796379774808884, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.43445590138435364, "reward_total_composite_std": 0.006304551847279072} {"timestamp_utc": "2026-04-13T12:25:47Z", "mode": "train", "global_step": 2067, "epoch": 0.20763435459568055, "loss": 0.0469, "grad_norm": 14.514013290405273, "learning_rate": 3.73939393939394e-06, "num_tokens": 3719274.0, "completions/mean_length": 35.0, "completions/min_length": 30.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.0, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.527634859085083, "rewards/meter/std": 0.30777183175086975, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9961965084075928, "rewards/repeat_soft/std": 0.0050888583064079285, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4008738696575165, "rewards/total_composite/std": 0.02962041273713112, "reward": 0.4008738696575165, "reward_std": 0.02962040901184082, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14774520695209503, "sampling/sampling_logp_difference/max": 2.5967350006103516, "sampling/importance_sampling_ratio/min": 0.07451647520065308, "sampling/importance_sampling_ratio/mean": 0.996990978717804, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6231069229543209, "clip_ratio/low_mean": 0.05282696336507797, "clip_ratio/low_min": 0.05282696336507797, "clip_ratio/high_mean": 0.03769909776747227, "clip_ratio/high_max": 0.03769909776747227, "clip_ratio/region_mean": 0.09052606113255024, "reward_total_mean": 0.4008738696575165, "reward_meter_mean": 0.527634859085083, "reward_meter_std": 0.30777183175086975, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9961965084075928, "reward_repeat_soft_std": 0.0050888583064079285, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4008738696575165, "reward_total_composite_std": 0.02962041273713112} {"timestamp_utc": "2026-04-13T12:25:55Z", "mode": "train", "global_step": 2068, "epoch": 0.20773480662983426, "loss": 0.048, "grad_norm": 7.030206203460693, "learning_rate": 3.736363636363637e-06, "num_tokens": 3721709.0, "completions/mean_length": 131.375, "completions/min_length": 121.0, "completions/max_length": 144.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.375, "completions/min_terminated_length": 121.0, "completions/max_terminated_length": 144.0, "rewards/meter/mean": 0.6101386547088623, "rewards/meter/std": 0.44752225279808044, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9160374402999878, "rewards/repeat_soft/std": 0.03248806670308113, "rewards/judge_quality/mean": 0.16500000655651093, "rewards/judge_quality/std": 0.02777460776269436, "rewards/total_composite/mean": 0.40356409549713135, "rewards/total_composite/std": 0.050754498690366745, "reward": 0.40356409549713135, "reward_std": 0.05075449496507645, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12449008226394653, "sampling/sampling_logp_difference/max": 4.021718502044678, "sampling/importance_sampling_ratio/min": 0.017922138795256615, "sampling/importance_sampling_ratio/mean": 1.0004775524139404, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.549030002206564, "clip_ratio/low_mean": 0.04585020337253809, "clip_ratio/low_min": 0.04585020337253809, "clip_ratio/high_mean": 0.07452918961644173, "clip_ratio/high_max": 0.07452918961644173, "clip_ratio/region_mean": 0.12037939298897982, "reward_total_mean": 0.40356409549713135, "reward_meter_mean": 0.6101386547088623, "reward_meter_std": 0.44752225279808044, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9160374402999878, "reward_repeat_soft_std": 0.03248806670308113, "reward_judge_quality_mean": 0.16500000655651093, "reward_judge_quality_std": 0.02777460776269436, "reward_total_composite_mean": 0.40356409549713135, "reward_total_composite_std": 0.050754498690366745} {"timestamp_utc": "2026-04-13T12:26:01Z", "mode": "train", "global_step": 2069, "epoch": 0.20783525866398794, "loss": 0.0619, "grad_norm": 9.696648597717285, "learning_rate": 3.7333333333333337e-06, "num_tokens": 3723350.0, "completions/mean_length": 47.125, "completions/min_length": 43.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.125, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.9494589567184448, "rewards/meter/std": 0.08970160037279129, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9439249038696289, "rewards/repeat_soft/std": 0.02473514713346958, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.434160977602005, "rewards/total_composite/std": 0.010766803286969662, "reward": 0.434160977602005, "reward_std": 0.010766806080937386, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11273051798343658, "sampling/sampling_logp_difference/max": 2.734816074371338, "sampling/importance_sampling_ratio/min": 0.06490594148635864, "sampling/importance_sampling_ratio/mean": 1.0210504531860352, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5686176270246506, "clip_ratio/low_mean": 0.030455291271209717, "clip_ratio/low_min": 0.030455291271209717, "clip_ratio/high_mean": 0.075993905775249, "clip_ratio/high_max": 0.075993905775249, "clip_ratio/region_mean": 0.10644919704645872, "reward_total_mean": 0.434160977602005, "reward_meter_mean": 0.9494589567184448, "reward_meter_std": 0.08970160037279129, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9439249038696289, "reward_repeat_soft_std": 0.02473514713346958, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.434160977602005, "reward_total_composite_std": 0.010766803286969662} {"timestamp_utc": "2026-04-13T12:26:07Z", "mode": "train", "global_step": 2070, "epoch": 0.20793571069814162, "loss": -0.0183, "grad_norm": 12.375938415527344, "learning_rate": 3.7303030303030306e-06, "num_tokens": 3724969.0, "completions/mean_length": 40.375, "completions/min_length": 35.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.375, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.8671349883079529, "rewards/meter/std": 0.3140276074409485, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9792593717575073, "rewards/repeat_soft/std": 0.020110072568058968, "rewards/judge_quality/mean": 0.1875, "rewards/judge_quality/std": 0.1060660108923912, "rewards/total_composite/mean": 0.45553040504455566, "rewards/total_composite/std": 0.07938668876886368, "reward": 0.45553040504455566, "reward_std": 0.07938668131828308, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11620399355888367, "sampling/sampling_logp_difference/max": 2.658374071121216, "sampling/importance_sampling_ratio/min": 0.0700620487332344, "sampling/importance_sampling_ratio/mean": 1.02402925491333, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7220793217420578, "clip_ratio/low_mean": 0.08675484010018408, "clip_ratio/low_min": 0.08675484010018408, "clip_ratio/high_mean": 0.02222222276031971, "clip_ratio/high_max": 0.02222222276031971, "clip_ratio/region_mean": 0.10897706286050379, "reward_total_mean": 0.45553040504455566, "reward_meter_mean": 0.8671349883079529, "reward_meter_std": 0.3140276074409485, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9792593717575073, "reward_repeat_soft_std": 0.020110072568058968, "reward_judge_quality_mean": 0.1875, "reward_judge_quality_std": 0.1060660108923912, "reward_total_composite_mean": 0.45553040504455566, "reward_total_composite_std": 0.07938668876886368} {"timestamp_utc": "2026-04-13T12:26:14Z", "mode": "train", "global_step": 2071, "epoch": 0.20803616273229533, "loss": -0.0811, "grad_norm": 11.69269847869873, "learning_rate": 3.727272727272728e-06, "num_tokens": 3726590.0, "completions/mean_length": 47.625, "completions/min_length": 37.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.625, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9296966791152954, "rewards/meter/std": 0.1094554215669632, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9387052655220032, "rewards/repeat_soft/std": 0.06813524663448334, "rewards/judge_quality/mean": 0.23250000178813934, "rewards/judge_quality/std": 0.1348809003829956, "rewards/total_composite/mean": 0.4744293987751007, "rewards/total_composite/std": 0.06318916380405426, "reward": 0.4744293987751007, "reward_std": 0.06318917125463486, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10637016594409943, "sampling/sampling_logp_difference/max": 1.755422592163086, "sampling/importance_sampling_ratio/min": 0.17283418774604797, "sampling/importance_sampling_ratio/mean": 0.9959914088249207, "sampling/importance_sampling_ratio/max": 1.7517482042312622, "entropy": 0.4855124205350876, "clip_ratio/low_mean": 0.07618559896945953, "clip_ratio/low_min": 0.07618559896945953, "clip_ratio/high_mean": 0.02797619067132473, "clip_ratio/high_max": 0.02797619067132473, "clip_ratio/region_mean": 0.10416178964078426, "reward_total_mean": 0.4744293987751007, "reward_meter_mean": 0.9296966791152954, "reward_meter_std": 0.1094554215669632, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9387052655220032, "reward_repeat_soft_std": 0.06813524663448334, "reward_judge_quality_mean": 0.23250000178813934, "reward_judge_quality_std": 0.1348809003829956, "reward_total_composite_mean": 0.4744293987751007, "reward_total_composite_std": 0.06318916380405426} {"timestamp_utc": "2026-04-13T12:26:21Z", "mode": "train", "global_step": 2072, "epoch": 0.208136614766449, "loss": 0.0753, "grad_norm": 8.115363121032715, "learning_rate": 3.7242424242424246e-06, "num_tokens": 3728520.0, "completions/mean_length": 90.25, "completions/min_length": 81.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.25, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.7392156720161438, "rewards/meter/std": 0.23921041190624237, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9616492986679077, "rewards/repeat_soft/std": 0.045542970299720764, "rewards/judge_quality/mean": 0.13875000178813934, "rewards/judge_quality/std": 0.022320719435811043, "rewards/total_composite/mean": 0.41035062074661255, "rewards/total_composite/std": 0.025785747915506363, "reward": 0.41035062074661255, "reward_std": 0.025785747915506363, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11131595075130463, "sampling/sampling_logp_difference/max": 1.636796474456787, "sampling/importance_sampling_ratio/min": 0.24769717454910278, "sampling/importance_sampling_ratio/mean": 1.0275774002075195, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7617958560585976, "clip_ratio/low_mean": 0.04378078877925873, "clip_ratio/low_min": 0.04378078877925873, "clip_ratio/high_mean": 0.06652565207332373, "clip_ratio/high_max": 0.06652565207332373, "clip_ratio/region_mean": 0.11030644085258245, "reward_total_mean": 0.41035062074661255, "reward_meter_mean": 0.7392156720161438, "reward_meter_std": 0.23921041190624237, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9616492986679077, "reward_repeat_soft_std": 0.045542970299720764, "reward_judge_quality_mean": 0.13875000178813934, "reward_judge_quality_std": 0.022320719435811043, "reward_total_composite_mean": 0.41035062074661255, "reward_total_composite_std": 0.025785747915506363} {"timestamp_utc": "2026-04-13T12:26:28Z", "mode": "train", "global_step": 2073, "epoch": 0.20823706680060272, "loss": 0.097, "grad_norm": 14.835057258605957, "learning_rate": 3.7212121212121215e-06, "num_tokens": 3730112.0, "completions/mean_length": 38.0, "completions/min_length": 28.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.0, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.9140802621841431, "rewards/meter/std": 0.09279971569776535, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8278565406799316, "rewards/repeat_soft/std": 0.16277727484703064, "rewards/judge_quality/mean": 0.16875001788139343, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.4241696000099182, "rewards/total_composite/std": 0.027534764260053635, "reward": 0.4241696000099182, "reward_std": 0.027534764260053635, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1308312863111496, "sampling/sampling_logp_difference/max": 2.29068922996521, "sampling/importance_sampling_ratio/min": 0.10119669139385223, "sampling/importance_sampling_ratio/mean": 0.9928392171859741, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.497556971386075, "clip_ratio/low_mean": 0.04509708611294627, "clip_ratio/low_min": 0.04509708611294627, "clip_ratio/high_mean": 0.0766675341874361, "clip_ratio/high_max": 0.0766675341874361, "clip_ratio/region_mean": 0.12176462030038238, "reward_total_mean": 0.4241696000099182, "reward_meter_mean": 0.9140802621841431, "reward_meter_std": 0.09279971569776535, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8278565406799316, "reward_repeat_soft_std": 0.16277727484703064, "reward_judge_quality_mean": 0.16875001788139343, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.4241696000099182, "reward_total_composite_std": 0.027534764260053635} {"timestamp_utc": "2026-04-13T12:26:36Z", "mode": "train", "global_step": 2074, "epoch": 0.2083375188347564, "loss": 0.0815, "grad_norm": 7.157658100128174, "learning_rate": 3.7181818181818187e-06, "num_tokens": 3732910.0, "completions/mean_length": 168.75, "completions/min_length": 145.0, "completions/max_length": 192.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 168.75, "completions/min_terminated_length": 145.0, "completions/max_terminated_length": 192.0, "rewards/meter/mean": 0.8324783444404602, "rewards/meter/std": 0.18502572178840637, "rewards/count_adherence/mean": 0.9583333134651184, "rewards/count_adherence/std": 0.07715168595314026, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7660987973213196, "rewards/repeat_soft/std": 0.03201473131775856, "rewards/judge_quality/mean": 0.16500000655651093, "rewards/judge_quality/std": 0.02777460776269436, "rewards/total_composite/mean": 0.3949919044971466, "rewards/total_composite/std": 0.03361143544316292, "reward": 0.3949919044971466, "reward_std": 0.03361142799258232, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09431398659944534, "sampling/sampling_logp_difference/max": 4.0692362785339355, "sampling/importance_sampling_ratio/min": 0.017090434208512306, "sampling/importance_sampling_ratio/mean": 1.0004268884658813, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.49766362458467484, "clip_ratio/low_mean": 0.025941302068531513, "clip_ratio/low_min": 0.025941302068531513, "clip_ratio/high_mean": 0.05744477640837431, "clip_ratio/high_max": 0.05744477640837431, "clip_ratio/region_mean": 0.08338607847690582, "reward_total_mean": 0.3949919044971466, "reward_meter_mean": 0.8324783444404602, "reward_meter_std": 0.18502572178840637, "reward_count_adherence_mean": 0.9583333134651184, "reward_count_adherence_std": 0.07715168595314026, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7660987973213196, "reward_repeat_soft_std": 0.03201473131775856, "reward_judge_quality_mean": 0.16500000655651093, "reward_judge_quality_std": 0.02777460776269436, "reward_total_composite_mean": 0.3949919044971466, "reward_total_composite_std": 0.03361143544316292} {"timestamp_utc": "2026-04-13T12:26:43Z", "mode": "train", "global_step": 2075, "epoch": 0.20843797086891008, "loss": 0.0437, "grad_norm": 19.99515151977539, "learning_rate": 3.7151515151515156e-06, "num_tokens": 3734292.0, "completions/mean_length": 22.75, "completions/min_length": 20.0, "completions/max_length": 27.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.75, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.8105221390724182, "rewards/meter/std": 0.35397836565971375, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9114496111869812, "rewards/repeat_soft/std": 0.06043412163853645, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.415743350982666, "rewards/total_composite/std": 0.04004393517971039, "reward": 0.415743350982666, "reward_std": 0.04004393890500069, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12008140236139297, "sampling/sampling_logp_difference/max": 1.7892513275146484, "sampling/importance_sampling_ratio/min": 0.1670852154493332, "sampling/importance_sampling_ratio/mean": 1.0087186098098755, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7540407590568066, "clip_ratio/low_mean": 0.03476190473884344, "clip_ratio/low_min": 0.03476190473884344, "clip_ratio/high_mean": 0.08422762854024768, "clip_ratio/high_max": 0.08422762854024768, "clip_ratio/region_mean": 0.11898953327909112, "reward_total_mean": 0.415743350982666, "reward_meter_mean": 0.8105221390724182, "reward_meter_std": 0.35397836565971375, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9114496111869812, "reward_repeat_soft_std": 0.06043412163853645, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.415743350982666, "reward_total_composite_std": 0.04004393517971039} {"timestamp_utc": "2026-04-13T12:26:50Z", "mode": "train", "global_step": 2076, "epoch": 0.2085384229030638, "loss": 0.1202, "grad_norm": 8.06759262084961, "learning_rate": 3.7121212121212124e-06, "num_tokens": 3736648.0, "completions/mean_length": 107.5, "completions/min_length": 82.0, "completions/max_length": 134.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.5, "completions/min_terminated_length": 82.0, "completions/max_terminated_length": 134.0, "rewards/meter/mean": 0.9633580446243286, "rewards/meter/std": 0.035794198513031006, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8788341283798218, "rewards/repeat_soft/std": 0.04080803319811821, "rewards/judge_quality/mean": 0.13875001668930054, "rewards/judge_quality/std": 0.02748376503586769, "rewards/total_composite/mean": 0.40023452043533325, "rewards/total_composite/std": 0.031872548162937164, "reward": 0.40023452043533325, "reward_std": 0.031872548162937164, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11378919333219528, "sampling/sampling_logp_difference/max": 1.6996946334838867, "sampling/importance_sampling_ratio/min": 0.18273931741714478, "sampling/importance_sampling_ratio/mean": 0.9930237531661987, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5079356506466866, "clip_ratio/low_mean": 0.04586605029180646, "clip_ratio/low_min": 0.04586605029180646, "clip_ratio/high_mean": 0.04101017955690622, "clip_ratio/high_max": 0.04101017955690622, "clip_ratio/region_mean": 0.08687622984871268, "reward_total_mean": 0.40023452043533325, "reward_meter_mean": 0.9633580446243286, "reward_meter_std": 0.035794198513031006, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8788341283798218, "reward_repeat_soft_std": 0.04080803319811821, "reward_judge_quality_mean": 0.13875001668930054, "reward_judge_quality_std": 0.02748376503586769, "reward_total_composite_mean": 0.40023452043533325, "reward_total_composite_std": 0.031872548162937164} {"timestamp_utc": "2026-04-13T12:26:56Z", "mode": "train", "global_step": 2077, "epoch": 0.20863887493721747, "loss": 0.0324, "grad_norm": 14.313187599182129, "learning_rate": 3.7090909090909092e-06, "num_tokens": 3738040.0, "completions/mean_length": 30.0, "completions/min_length": 27.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.0, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.6820467114448547, "rewards/meter/std": 0.23005546629428864, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9629801511764526, "rewards/repeat_soft/std": 0.0550442710518837, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4109466075897217, "rewards/total_composite/std": 0.025824129581451416, "reward": 0.4109466075897217, "reward_std": 0.025824127718806267, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08204793930053711, "sampling/sampling_logp_difference/max": 1.4037678241729736, "sampling/importance_sampling_ratio/min": 0.24566958844661713, "sampling/importance_sampling_ratio/mean": 1.0015180110931396, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4720030911266804, "clip_ratio/low_mean": 0.012449766509234905, "clip_ratio/low_min": 0.012449766509234905, "clip_ratio/high_mean": 0.051067705266177654, "clip_ratio/high_max": 0.051067705266177654, "clip_ratio/region_mean": 0.06351747177541256, "reward_total_mean": 0.4109466075897217, "reward_meter_mean": 0.6820467114448547, "reward_meter_std": 0.23005546629428864, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9629801511764526, "reward_repeat_soft_std": 0.0550442710518837, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4109466075897217, "reward_total_composite_std": 0.025824129581451416} {"timestamp_utc": "2026-04-13T12:27:04Z", "mode": "train", "global_step": 2078, "epoch": 0.20873932697137118, "loss": 0.1028, "grad_norm": 6.437441825866699, "learning_rate": 3.7060606060606065e-06, "num_tokens": 3740218.0, "completions/mean_length": 101.25, "completions/min_length": 87.0, "completions/max_length": 117.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 101.25, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.9617483615875244, "rewards/meter/std": 0.0327635258436203, "rewards/count_adherence/mean": 0.84375, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.809037983417511, "rewards/repeat_soft/std": 0.07080276310443878, "rewards/judge_quality/mean": 0.1274999976158142, "rewards/judge_quality/std": 0.02121320739388466, "rewards/total_composite/mean": 0.36988019943237305, "rewards/total_composite/std": 0.03989952802658081, "reward": 0.36988019943237305, "reward_std": 0.03989953175187111, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09450139850378036, "sampling/sampling_logp_difference/max": 2.4037623405456543, "sampling/importance_sampling_ratio/min": 0.09037727862596512, "sampling/importance_sampling_ratio/mean": 1.0028011798858643, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.41972821950912476, "clip_ratio/low_mean": 0.05465828627347946, "clip_ratio/low_min": 0.05465828627347946, "clip_ratio/high_mean": 0.02143743447959423, "clip_ratio/high_max": 0.02143743447959423, "clip_ratio/region_mean": 0.07609572075307369, "reward_total_mean": 0.36988019943237305, "reward_meter_mean": 0.9617483615875244, "reward_meter_std": 0.0327635258436203, "reward_count_adherence_mean": 0.84375, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.809037983417511, "reward_repeat_soft_std": 0.07080276310443878, "reward_judge_quality_mean": 0.1274999976158142, "reward_judge_quality_std": 0.02121320739388466, "reward_total_composite_mean": 0.36988019943237305, "reward_total_composite_std": 0.03989952802658081} {"timestamp_utc": "2026-04-13T12:27:11Z", "mode": "train", "global_step": 2079, "epoch": 0.20883977900552486, "loss": -0.0068, "grad_norm": 14.443037986755371, "learning_rate": 3.7030303030303033e-06, "num_tokens": 3741818.0, "completions/mean_length": 52.0, "completions/min_length": 49.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.0, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9573231935501099, "rewards/meter/std": 0.07338250428438187, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.975223183631897, "rewards/repeat_soft/std": 0.013552870601415634, "rewards/judge_quality/mean": 0.1574999988079071, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.44444313645362854, "rewards/total_composite/std": 0.011920580640435219, "reward": 0.44444313645362854, "reward_std": 0.011920583434402943, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08642370253801346, "sampling/sampling_logp_difference/max": 1.1978371143341064, "sampling/importance_sampling_ratio/min": 0.30184635519981384, "sampling/importance_sampling_ratio/mean": 1.0158339738845825, "sampling/importance_sampling_ratio/max": 1.9727880954742432, "entropy": 0.5891767293214798, "clip_ratio/low_mean": 0.048753238283097744, "clip_ratio/low_min": 0.048753238283097744, "clip_ratio/high_mean": 0.04219135455787182, "clip_ratio/high_max": 0.04219135455787182, "clip_ratio/region_mean": 0.09094459284096956, "reward_total_mean": 0.44444313645362854, "reward_meter_mean": 0.9573231935501099, "reward_meter_std": 0.07338250428438187, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.975223183631897, "reward_repeat_soft_std": 0.013552870601415634, "reward_judge_quality_mean": 0.1574999988079071, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.44444313645362854, "reward_total_composite_std": 0.011920580640435219} {"timestamp_utc": "2026-04-13T12:27:22Z", "mode": "train", "global_step": 2080, "epoch": 0.20894023103967854, "loss": -0.2059, "grad_norm": 1.7132174968719482, "learning_rate": 3.7e-06, "num_tokens": 3744129.0, "completions/mean_length": 161.875, "completions/min_length": 93.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 111.85714721679688, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 137.0, "rewards/meter/mean": 0.7818764448165894, "rewards/meter/std": 0.308095246553421, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.2651650309562683, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8692042231559753, "rewards/repeat_soft/std": 0.05775836110115051, "rewards/judge_quality/mean": 0.14125001430511475, "rewards/judge_quality/std": 0.04733996093273163, "rewards/total_composite/mean": 0.36371177434921265, "rewards/total_composite/std": 0.14940164983272552, "reward": 0.36371177434921265, "reward_std": 0.14940164983272552, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09953910112380981, "sampling/sampling_logp_difference/max": 1.7864031791687012, "sampling/importance_sampling_ratio/min": 0.16756176948547363, "sampling/importance_sampling_ratio/mean": 0.9994521141052246, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.49852072820067406, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.08590791467577219, "clip_ratio/high_max": 0.08590791467577219, "clip_ratio/region_mean": 0.08590791467577219, "reward_total_mean": 0.36371177434921265, "reward_meter_mean": 0.7818764448165894, "reward_meter_std": 0.308095246553421, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.2651650309562683, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8692042231559753, "reward_repeat_soft_std": 0.05775836110115051, "reward_judge_quality_mean": 0.14125001430511475, "reward_judge_quality_std": 0.04733996093273163, "reward_total_composite_mean": 0.36371177434921265, "reward_total_composite_std": 0.14940164983272552} {"timestamp_utc": "2026-04-13T12:27:34Z", "mode": "train", "global_step": 2081, "epoch": 0.20904068307383225, "loss": -0.1162, "grad_norm": 2.2427473068237305, "learning_rate": 3.6969696969696974e-06, "num_tokens": 3745643.0, "completions/mean_length": 95.25, "completions/min_length": 26.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 35.71428680419922, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.4832600951194763, "rewards/meter/std": 0.4000207781791687, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9759777784347534, "rewards/repeat_soft/std": 0.023617975413799286, "rewards/judge_quality/mean": 0.13750000298023224, "rewards/judge_quality/std": 0.0353553406894207, "rewards/total_composite/mean": 0.3494424819946289, "rewards/total_composite/std": 0.14582893252372742, "reward": 0.3494424819946289, "reward_std": 0.14582891762256622, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16125179827213287, "sampling/sampling_logp_difference/max": 2.206164836883545, "sampling/importance_sampling_ratio/min": 0.1101221814751625, "sampling/importance_sampling_ratio/mean": 1.00516676902771, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6992188319563866, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.1526223225519061, "clip_ratio/high_max": 0.1526223225519061, "clip_ratio/region_mean": 0.1526223225519061, "reward_total_mean": 0.3494424819946289, "reward_meter_mean": 0.4832600951194763, "reward_meter_std": 0.4000207781791687, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9759777784347534, "reward_repeat_soft_std": 0.023617975413799286, "reward_judge_quality_mean": 0.13750000298023224, "reward_judge_quality_std": 0.0353553406894207, "reward_total_composite_mean": 0.3494424819946289, "reward_total_composite_std": 0.14582893252372742} {"timestamp_utc": "2026-04-13T12:27:40Z", "mode": "train", "global_step": 2082, "epoch": 0.20914113510798593, "loss": 0.0459, "grad_norm": 12.529629707336426, "learning_rate": 3.6939393939393942e-06, "num_tokens": 3747157.0, "completions/mean_length": 46.25, "completions/min_length": 38.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.25, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.4912826418876648, "rewards/meter/std": 0.30648428201675415, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9786837100982666, "rewards/repeat_soft/std": 0.022699061781167984, "rewards/judge_quality/mean": 0.19499999284744263, "rewards/judge_quality/std": 0.10392304509878159, "rewards/total_composite/mean": 0.4009936451911926, "rewards/total_composite/std": 0.03267580643296242, "reward": 0.4009936451911926, "reward_std": 0.03267580270767212, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12824684381484985, "sampling/sampling_logp_difference/max": 2.061049699783325, "sampling/importance_sampling_ratio/min": 0.12732024490833282, "sampling/importance_sampling_ratio/mean": 1.009272575378418, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7472609356045723, "clip_ratio/low_mean": 0.08138155192136765, "clip_ratio/low_min": 0.08138155192136765, "clip_ratio/high_mean": 0.04700176510959864, "clip_ratio/high_max": 0.04700176510959864, "clip_ratio/region_mean": 0.12838331703096628, "reward_total_mean": 0.4009936451911926, "reward_meter_mean": 0.4912826418876648, "reward_meter_std": 0.30648428201675415, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9786837100982666, "reward_repeat_soft_std": 0.022699061781167984, "reward_judge_quality_mean": 0.19499999284744263, "reward_judge_quality_std": 0.10392304509878159, "reward_total_composite_mean": 0.4009936451911926, "reward_total_composite_std": 0.03267580643296242} {"timestamp_utc": "2026-04-13T12:27:47Z", "mode": "train", "global_step": 2083, "epoch": 0.20924158714213964, "loss": 0.0185, "grad_norm": 14.358192443847656, "learning_rate": 3.690909090909091e-06, "num_tokens": 3748664.0, "completions/mean_length": 44.375, "completions/min_length": 39.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.375, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.6349843740463257, "rewards/meter/std": 0.42381957173347473, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9558464288711548, "rewards/repeat_soft/std": 0.032159652560949326, "rewards/judge_quality/mean": 0.14625000953674316, "rewards/judge_quality/std": 0.010606604628264904, "rewards/total_composite/mean": 0.4029209315776825, "rewards/total_composite/std": 0.04163876920938492, "reward": 0.4029209315776825, "reward_std": 0.04163876920938492, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11028869450092316, "sampling/sampling_logp_difference/max": 1.2160595655441284, "sampling/importance_sampling_ratio/min": 0.29639580845832825, "sampling/importance_sampling_ratio/mean": 1.0181822776794434, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7787529304623604, "clip_ratio/low_mean": 0.04384576342999935, "clip_ratio/low_min": 0.04384576342999935, "clip_ratio/high_mean": 0.08037643367424607, "clip_ratio/high_max": 0.08037643367424607, "clip_ratio/region_mean": 0.12422219710424542, "reward_total_mean": 0.4029209315776825, "reward_meter_mean": 0.6349843740463257, "reward_meter_std": 0.42381957173347473, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9558464288711548, "reward_repeat_soft_std": 0.032159652560949326, "reward_judge_quality_mean": 0.14625000953674316, "reward_judge_quality_std": 0.010606604628264904, "reward_total_composite_mean": 0.4029209315776825, "reward_total_composite_std": 0.04163876920938492} {"timestamp_utc": "2026-04-13T12:27:54Z", "mode": "train", "global_step": 2084, "epoch": 0.20934203917629332, "loss": 0.1051, "grad_norm": 8.194664001464844, "learning_rate": 3.687878787878788e-06, "num_tokens": 3751053.0, "completions/mean_length": 106.625, "completions/min_length": 97.0, "completions/max_length": 140.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.625, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 140.0, "rewards/meter/mean": 0.8641301393508911, "rewards/meter/std": 0.2113206684589386, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8112736344337463, "rewards/repeat_soft/std": 0.03796151652932167, "rewards/judge_quality/mean": 0.1537500023841858, "rewards/judge_quality/std": 0.029730942100286484, "rewards/total_composite/mean": 0.40080052614212036, "rewards/total_composite/std": 0.02927142195403576, "reward": 0.40080052614212036, "reward_std": 0.02927141822874546, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08997472375631332, "sampling/sampling_logp_difference/max": 2.6632847785949707, "sampling/importance_sampling_ratio/min": 0.06971883028745651, "sampling/importance_sampling_ratio/mean": 1.003967523574829, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.46710870042443275, "clip_ratio/low_mean": 0.03522890596650541, "clip_ratio/low_min": 0.03522890596650541, "clip_ratio/high_mean": 0.03521409630775452, "clip_ratio/high_max": 0.03521409630775452, "clip_ratio/region_mean": 0.07044300227425992, "reward_total_mean": 0.40080052614212036, "reward_meter_mean": 0.8641301393508911, "reward_meter_std": 0.2113206684589386, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8112736344337463, "reward_repeat_soft_std": 0.03796151652932167, "reward_judge_quality_mean": 0.1537500023841858, "reward_judge_quality_std": 0.029730942100286484, "reward_total_composite_mean": 0.40080052614212036, "reward_total_composite_std": 0.02927142195403576} {"timestamp_utc": "2026-04-13T12:28:01Z", "mode": "train", "global_step": 2085, "epoch": 0.209442491210447, "loss": 0.0659, "grad_norm": 12.54304027557373, "learning_rate": 3.684848484848485e-06, "num_tokens": 3752582.0, "completions/mean_length": 41.125, "completions/min_length": 32.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.125, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.8170626163482666, "rewards/meter/std": 0.3091009855270386, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8963360786437988, "rewards/repeat_soft/std": 0.07029309868812561, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.41411399841308594, "rewards/total_composite/std": 0.029479196295142174, "reward": 0.41411399841308594, "reward_std": 0.029479194432497025, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11653422564268112, "sampling/sampling_logp_difference/max": 1.3648093938827515, "sampling/importance_sampling_ratio/min": 0.25542935729026794, "sampling/importance_sampling_ratio/mean": 1.0088908672332764, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6153091266751289, "clip_ratio/low_mean": 0.03251900803297758, "clip_ratio/low_min": 0.03251900803297758, "clip_ratio/high_mean": 0.05771985976025462, "clip_ratio/high_max": 0.05771985976025462, "clip_ratio/region_mean": 0.0902388677932322, "reward_total_mean": 0.41411399841308594, "reward_meter_mean": 0.8170626163482666, "reward_meter_std": 0.3091009855270386, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8963360786437988, "reward_repeat_soft_std": 0.07029309868812561, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.41411399841308594, "reward_total_composite_std": 0.029479196295142174} {"timestamp_utc": "2026-04-13T12:28:13Z", "mode": "train", "global_step": 2086, "epoch": 0.2095429432446007, "loss": -0.1198, "grad_norm": 1.476021409034729, "learning_rate": 3.681818181818182e-06, "num_tokens": 3754249.0, "completions/mean_length": 165.375, "completions/min_length": 43.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 49.833335876464844, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.7697263956069946, "rewards/meter/std": 0.23756560683250427, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9250380992889404, "rewards/repeat_soft/std": 0.07582405209541321, "rewards/judge_quality/mean": 0.13625000417232513, "rewards/judge_quality/std": 0.057305578142404556, "rewards/total_composite/mean": 0.32509738206863403, "rewards/total_composite/std": 0.20268148183822632, "reward": 0.32509738206863403, "reward_std": 0.20268146693706512, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0769723579287529, "sampling/sampling_logp_difference/max": 2.3819503784179688, "sampling/importance_sampling_ratio/min": 0.09237024188041687, "sampling/importance_sampling_ratio/mean": 1.0031273365020752, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2697983533143997, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.050510191125795245, "clip_ratio/high_max": 0.050510191125795245, "clip_ratio/region_mean": 0.050510191125795245, "reward_total_mean": 0.32509738206863403, "reward_meter_mean": 0.7697263956069946, "reward_meter_std": 0.23756560683250427, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9250380992889404, "reward_repeat_soft_std": 0.07582405209541321, "reward_judge_quality_mean": 0.13625000417232513, "reward_judge_quality_std": 0.057305578142404556, "reward_total_composite_mean": 0.32509738206863403, "reward_total_composite_std": 0.20268148183822632} {"timestamp_utc": "2026-04-13T12:28:20Z", "mode": "train", "global_step": 2087, "epoch": 0.2096433952787544, "loss": 0.0692, "grad_norm": 16.328943252563477, "learning_rate": 3.678787878787879e-06, "num_tokens": 3755826.0, "completions/mean_length": 37.125, "completions/min_length": 30.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.125, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.8387521505355835, "rewards/meter/std": 0.19725178182125092, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9575532078742981, "rewards/repeat_soft/std": 0.03129439800977707, "rewards/judge_quality/mean": 0.16124999523162842, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.43037697672843933, "rewards/total_composite/std": 0.02135138213634491, "reward": 0.43037697672843933, "reward_std": 0.02135137841105461, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08730937540531158, "sampling/sampling_logp_difference/max": 2.1456055641174316, "sampling/importance_sampling_ratio/min": 0.11699716001749039, "sampling/importance_sampling_ratio/mean": 0.9835996031761169, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3454021457582712, "clip_ratio/low_mean": 0.006756756920367479, "clip_ratio/low_min": 0.006756756920367479, "clip_ratio/high_mean": 0.080037297680974, "clip_ratio/high_max": 0.080037297680974, "clip_ratio/region_mean": 0.08679405460134149, "reward_total_mean": 0.43037697672843933, "reward_meter_mean": 0.8387521505355835, "reward_meter_std": 0.19725178182125092, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9575532078742981, "reward_repeat_soft_std": 0.03129439800977707, "reward_judge_quality_mean": 0.16124999523162842, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.43037697672843933, "reward_total_composite_std": 0.02135138213634491} {"timestamp_utc": "2026-04-13T12:28:27Z", "mode": "train", "global_step": 2088, "epoch": 0.2097438473129081, "loss": 0.0317, "grad_norm": 11.181440353393555, "learning_rate": 3.6757575757575757e-06, "num_tokens": 3757367.0, "completions/mean_length": 36.625, "completions/min_length": 35.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.625, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.6886870861053467, "rewards/meter/std": 0.36762264370918274, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9336532354354858, "rewards/repeat_soft/std": 0.053609877824783325, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.40719497203826904, "rewards/total_composite/std": 0.03439675644040108, "reward": 0.40719497203826904, "reward_std": 0.034396763890981674, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09574722498655319, "sampling/sampling_logp_difference/max": 1.711014747619629, "sampling/importance_sampling_ratio/min": 0.18068234622478485, "sampling/importance_sampling_ratio/mean": 1.0201098918914795, "sampling/importance_sampling_ratio/max": 1.901428461074829, "entropy": 0.6929846182465553, "clip_ratio/low_mean": 0.03128175204619765, "clip_ratio/low_min": 0.03128175204619765, "clip_ratio/high_mean": 0.06482612853869796, "clip_ratio/high_max": 0.06482612853869796, "clip_ratio/region_mean": 0.09610788058489561, "reward_total_mean": 0.40719497203826904, "reward_meter_mean": 0.6886870861053467, "reward_meter_std": 0.36762264370918274, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9336532354354858, "reward_repeat_soft_std": 0.053609877824783325, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.40719497203826904, "reward_total_composite_std": 0.03439675644040108} {"timestamp_utc": "2026-04-13T12:28:45Z", "mode": "train", "global_step": 2089, "epoch": 0.20984429934706178, "loss": -0.1467, "grad_norm": 1.9923795461654663, "learning_rate": 3.672727272727273e-06, "num_tokens": 3759089.0, "completions/mean_length": 116.25, "completions/min_length": 46.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 59.71428680419922, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.5514540672302246, "rewards/meter/std": 0.40759724378585815, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9403271079063416, "rewards/repeat_soft/std": 0.03892381489276886, "rewards/judge_quality/mean": 0.13375000655651093, "rewards/judge_quality/std": 0.03889087587594986, "rewards/total_composite/mean": 0.3352453112602234, "rewards/total_composite/std": 0.1382029503583908, "reward": 0.3352453112602234, "reward_std": 0.1382029503583908, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10336823761463165, "sampling/sampling_logp_difference/max": 1.3734700679779053, "sampling/importance_sampling_ratio/min": 0.25322672724723816, "sampling/importance_sampling_ratio/mean": 1.0265514850616455, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5914758183062077, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.07886998308822513, "clip_ratio/high_max": 0.07886998308822513, "clip_ratio/region_mean": 0.07886998308822513, "reward_total_mean": 0.3352453112602234, "reward_meter_mean": 0.5514540672302246, "reward_meter_std": 0.40759724378585815, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9403271079063416, "reward_repeat_soft_std": 0.03892381489276886, "reward_judge_quality_mean": 0.13375000655651093, "reward_judge_quality_std": 0.03889087587594986, "reward_total_composite_mean": 0.3352453112602234, "reward_total_composite_std": 0.1382029503583908} {"timestamp_utc": "2026-04-13T12:28:53Z", "mode": "train", "global_step": 2090, "epoch": 0.20994475138121546, "loss": -0.0396, "grad_norm": 8.443550109863281, "learning_rate": 3.6696969696969697e-06, "num_tokens": 3761106.0, "completions/mean_length": 80.125, "completions/min_length": 62.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.125, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.616234540939331, "rewards/meter/std": 0.30755528807640076, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8544143438339233, "rewards/repeat_soft/std": 0.059049081057310104, "rewards/judge_quality/mean": 0.16875001788139343, "rewards/judge_quality/std": 0.022320719435811043, "rewards/total_composite/mean": 0.3969932198524475, "rewards/total_composite/std": 0.03428816422820091, "reward": 0.3969932198524475, "reward_std": 0.034288160502910614, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10842517018318176, "sampling/sampling_logp_difference/max": 1.542647361755371, "sampling/importance_sampling_ratio/min": 0.23686398565769196, "sampling/importance_sampling_ratio/mean": 1.0151734352111816, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7614212185144424, "clip_ratio/low_mean": 0.04624854167923331, "clip_ratio/low_min": 0.04624854167923331, "clip_ratio/high_mean": 0.05727146123535931, "clip_ratio/high_max": 0.05727146123535931, "clip_ratio/region_mean": 0.10352000291459262, "reward_total_mean": 0.3969932198524475, "reward_meter_mean": 0.616234540939331, "reward_meter_std": 0.30755528807640076, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8544143438339233, "reward_repeat_soft_std": 0.059049081057310104, "reward_judge_quality_mean": 0.16875001788139343, "reward_judge_quality_std": 0.022320719435811043, "reward_total_composite_mean": 0.3969932198524475, "reward_total_composite_std": 0.03428816422820091} {"timestamp_utc": "2026-04-13T12:28:59Z", "mode": "train", "global_step": 2091, "epoch": 0.21004520341536917, "loss": 0.0817, "grad_norm": 14.345635414123535, "learning_rate": 3.6666666666666666e-06, "num_tokens": 3762567.0, "completions/mean_length": 36.625, "completions/min_length": 32.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.625, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.9297787547111511, "rewards/meter/std": 0.13060350716114044, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9389592409133911, "rewards/repeat_soft/std": 0.05289275944232941, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.43149733543395996, "rewards/total_composite/std": 0.015238284133374691, "reward": 0.43149733543395996, "reward_std": 0.015238286927342415, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13995887339115143, "sampling/sampling_logp_difference/max": 1.4572899341583252, "sampling/importance_sampling_ratio/min": 0.24173982441425323, "sampling/importance_sampling_ratio/mean": 0.9981912970542908, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7522866874933243, "clip_ratio/low_mean": 0.058080810122191906, "clip_ratio/low_min": 0.058080810122191906, "clip_ratio/high_mean": 0.10172371473163366, "clip_ratio/high_max": 0.10172371473163366, "clip_ratio/region_mean": 0.15980452485382557, "reward_total_mean": 0.43149733543395996, "reward_meter_mean": 0.9297787547111511, "reward_meter_std": 0.13060350716114044, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9389592409133911, "reward_repeat_soft_std": 0.05289275944232941, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.43149733543395996, "reward_total_composite_std": 0.015238284133374691} {"timestamp_utc": "2026-04-13T12:29:05Z", "mode": "train", "global_step": 2092, "epoch": 0.21014565544952285, "loss": -0.0283, "grad_norm": 15.965852737426758, "learning_rate": 3.6636363636363643e-06, "num_tokens": 3763992.0, "completions/mean_length": 31.125, "completions/min_length": 27.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.125, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.8399671316146851, "rewards/meter/std": 0.24304862320423126, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9808377623558044, "rewards/repeat_soft/std": 0.020741088315844536, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4290224611759186, "rewards/total_composite/std": 0.02329912595450878, "reward": 0.4290224611759186, "reward_std": 0.02329912595450878, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08482443541288376, "sampling/sampling_logp_difference/max": 1.498971700668335, "sampling/importance_sampling_ratio/min": 0.22335973381996155, "sampling/importance_sampling_ratio/mean": 1.0098867416381836, "sampling/importance_sampling_ratio/max": 1.4488673210144043, "entropy": 0.4668281860649586, "clip_ratio/low_mean": 0.025388291105628014, "clip_ratio/low_min": 0.025388291105628014, "clip_ratio/high_mean": 0.07872221665456891, "clip_ratio/high_max": 0.07872221665456891, "clip_ratio/region_mean": 0.10411050776019692, "reward_total_mean": 0.4290224611759186, "reward_meter_mean": 0.8399671316146851, "reward_meter_std": 0.24304862320423126, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9808377623558044, "reward_repeat_soft_std": 0.020741088315844536, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4290224611759186, "reward_total_composite_std": 0.02329912595450878} {"timestamp_utc": "2026-04-13T12:29:12Z", "mode": "train", "global_step": 2093, "epoch": 0.21024610748367653, "loss": -0.0161, "grad_norm": 10.055636405944824, "learning_rate": 3.660606060606061e-06, "num_tokens": 3766114.0, "completions/mean_length": 76.25, "completions/min_length": 67.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.25, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.8110104203224182, "rewards/meter/std": 0.3234376907348633, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8537755012512207, "rewards/repeat_soft/std": 0.04971456155180931, "rewards/judge_quality/mean": 0.18000000715255737, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4229545593261719, "rewards/total_composite/std": 0.03832192346453667, "reward": 0.4229545593261719, "reward_std": 0.03832192346453667, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0807785764336586, "sampling/sampling_logp_difference/max": 1.7626324892044067, "sampling/importance_sampling_ratio/min": 0.17159254848957062, "sampling/importance_sampling_ratio/mean": 1.0127191543579102, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4079204872250557, "clip_ratio/low_mean": 0.023201302159577608, "clip_ratio/low_min": 0.023201302159577608, "clip_ratio/high_mean": 0.0597048276104033, "clip_ratio/high_max": 0.0597048276104033, "clip_ratio/region_mean": 0.08290612976998091, "reward_total_mean": 0.4229545593261719, "reward_meter_mean": 0.8110104203224182, "reward_meter_std": 0.3234376907348633, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8537755012512207, "reward_repeat_soft_std": 0.04971456155180931, "reward_judge_quality_mean": 0.18000000715255737, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4229545593261719, "reward_total_composite_std": 0.03832192346453667} {"timestamp_utc": "2026-04-13T12:29:18Z", "mode": "train", "global_step": 2094, "epoch": 0.21034655951783024, "loss": 0.0236, "grad_norm": 15.249076843261719, "learning_rate": 3.657575757575758e-06, "num_tokens": 3767484.0, "completions/mean_length": 19.25, "completions/min_length": 18.0, "completions/max_length": 21.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 19.25, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 21.0, "rewards/meter/mean": 0.9565207958221436, "rewards/meter/std": 0.07130192965269089, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8878955841064453, "rewards/repeat_soft/std": 0.1174720972776413, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4264451265335083, "rewards/total_composite/std": 0.01699831150472164, "reward": 0.4264451265335083, "reward_std": 0.016998320817947388, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09562919288873672, "sampling/sampling_logp_difference/max": 0.7266945838928223, "sampling/importance_sampling_ratio/min": 0.5248639583587646, "sampling/importance_sampling_ratio/mean": 1.044858694076538, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.764565996825695, "clip_ratio/low_mean": 0.04444444505497813, "clip_ratio/low_min": 0.04444444505497813, "clip_ratio/high_mean": 0.06150793796405196, "clip_ratio/high_max": 0.06150793796405196, "clip_ratio/region_mean": 0.1059523830190301, "reward_total_mean": 0.4264451265335083, "reward_meter_mean": 0.9565207958221436, "reward_meter_std": 0.07130192965269089, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8878955841064453, "reward_repeat_soft_std": 0.1174720972776413, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4264451265335083, "reward_total_composite_std": 0.01699831150472164} {"timestamp_utc": "2026-04-13T12:29:29Z", "mode": "train", "global_step": 2095, "epoch": 0.21044701155198392, "loss": -0.1317, "grad_norm": 1.4918196201324463, "learning_rate": 3.654545454545455e-06, "num_tokens": 3769136.0, "completions/mean_length": 235.5, "completions/min_length": 58.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 69.5999984741211, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.6806774139404297, "rewards/meter/std": 0.41470572352409363, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.2357022762298584, "rewards/hard_gate/mean": 0.625, "rewards/hard_gate/std": 0.5175492167472839, "rewards/repeat_soft/mean": 0.9118834733963013, "rewards/repeat_soft/std": 0.05751105025410652, "rewards/judge_quality/mean": 0.10875000059604645, "rewards/judge_quality/std": 0.052218638360500336, "rewards/total_composite/mean": 0.26164764165878296, "rewards/total_composite/std": 0.21728645265102386, "reward": 0.26164764165878296, "reward_std": 0.21728643774986267, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10722679644823074, "sampling/sampling_logp_difference/max": 2.4546332359313965, "sampling/importance_sampling_ratio/min": 0.08589468896389008, "sampling/importance_sampling_ratio/mean": 1.0040150880813599, "sampling/importance_sampling_ratio/max": 1.8782011270523071, "entropy": 0.3790879473090172, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.06049556843936443, "clip_ratio/high_max": 0.06049556843936443, "clip_ratio/region_mean": 0.06049556843936443, "reward_total_mean": 0.26164764165878296, "reward_meter_mean": 0.6806774139404297, "reward_meter_std": 0.41470572352409363, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.2357022762298584, "reward_hard_gate_mean": 0.625, "reward_hard_gate_std": 0.5175492167472839, "reward_repeat_soft_mean": 0.9118834733963013, "reward_repeat_soft_std": 0.05751105025410652, "reward_judge_quality_mean": 0.10875000059604645, "reward_judge_quality_std": 0.052218638360500336, "reward_total_composite_mean": 0.26164764165878296, "reward_total_composite_std": 0.21728645265102386} {"timestamp_utc": "2026-04-13T12:29:35Z", "mode": "train", "global_step": 2096, "epoch": 0.21054746358613763, "loss": 0.0259, "grad_norm": 11.645296096801758, "learning_rate": 3.651515151515152e-06, "num_tokens": 3770571.0, "completions/mean_length": 22.375, "completions/min_length": 22.0, "completions/max_length": 25.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.375, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 25.0, "rewards/meter/mean": 0.8985494375228882, "rewards/meter/std": 0.16152767837047577, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8453850746154785, "rewards/repeat_soft/std": 0.1095879003405571, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4144163429737091, "rewards/total_composite/std": 0.013869404792785645, "reward": 0.4144163429737091, "reward_std": 0.013869399204850197, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0835302397608757, "sampling/sampling_logp_difference/max": 0.940463662147522, "sampling/importance_sampling_ratio/min": 0.3904467821121216, "sampling/importance_sampling_ratio/mean": 1.0118560791015625, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5201849527657032, "clip_ratio/low_mean": 0.039090909995138645, "clip_ratio/low_min": 0.039090909995138645, "clip_ratio/high_mean": 0.011363636702299118, "clip_ratio/high_max": 0.011363636702299118, "clip_ratio/region_mean": 0.05045454669743776, "reward_total_mean": 0.4144163429737091, "reward_meter_mean": 0.8985494375228882, "reward_meter_std": 0.16152767837047577, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8453850746154785, "reward_repeat_soft_std": 0.1095879003405571, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4144163429737091, "reward_total_composite_std": 0.013869404792785645} {"timestamp_utc": "2026-04-13T12:29:47Z", "mode": "train", "global_step": 2097, "epoch": 0.2106479156202913, "loss": -0.1006, "grad_norm": 1.2347784042358398, "learning_rate": 3.648484848484849e-06, "num_tokens": 3771978.0, "completions/mean_length": 158.875, "completions/min_length": 36.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 41.16666793823242, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.9143640995025635, "rewards/meter/std": 0.12278993427753448, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9353083968162537, "rewards/repeat_soft/std": 0.037512220442295074, "rewards/judge_quality/mean": 0.125, "rewards/judge_quality/std": 0.046291008591651917, "rewards/total_composite/mean": 0.3247089385986328, "rewards/total_composite/std": 0.2005159854888916, "reward": 0.3247089385986328, "reward_std": 0.2005159854888916, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1059136688709259, "sampling/sampling_logp_difference/max": 1.8650531768798828, "sampling/importance_sampling_ratio/min": 0.15488797426223755, "sampling/importance_sampling_ratio/mean": 0.9949530363082886, "sampling/importance_sampling_ratio/max": 1.6144154071807861, "entropy": 0.442423515021801, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.08202262991108, "clip_ratio/high_max": 0.08202262991108, "clip_ratio/region_mean": 0.08202262991108, "reward_total_mean": 0.3247089385986328, "reward_meter_mean": 0.9143640995025635, "reward_meter_std": 0.12278993427753448, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9353083968162537, "reward_repeat_soft_std": 0.037512220442295074, "reward_judge_quality_mean": 0.125, "reward_judge_quality_std": 0.046291008591651917, "reward_total_composite_mean": 0.3247089385986328, "reward_total_composite_std": 0.2005159854888916} {"timestamp_utc": "2026-04-13T12:29:58Z", "mode": "train", "global_step": 2098, "epoch": 0.210748367654445, "loss": -0.1383, "grad_norm": 1.7239394187927246, "learning_rate": 3.645454545454546e-06, "num_tokens": 3773626.0, "completions/mean_length": 106.0, "completions/min_length": 41.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 48.000003814697266, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.8133279085159302, "rewards/meter/std": 0.29564806818962097, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9431638717651367, "rewards/repeat_soft/std": 0.037652309983968735, "rewards/judge_quality/mean": 0.14124999940395355, "rewards/judge_quality/std": 0.038335926830768585, "rewards/total_composite/mean": 0.37816929817199707, "rewards/total_composite/std": 0.15340575575828552, "reward": 0.37816929817199707, "reward_std": 0.15340575575828552, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09945490211248398, "sampling/sampling_logp_difference/max": 2.006791114807129, "sampling/importance_sampling_ratio/min": 0.13441932201385498, "sampling/importance_sampling_ratio/mean": 1.0164374113082886, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.554730236530304, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.09761778451502323, "clip_ratio/high_max": 0.09761778451502323, "clip_ratio/region_mean": 0.09761778451502323, "reward_total_mean": 0.37816929817199707, "reward_meter_mean": 0.8133279085159302, "reward_meter_std": 0.29564806818962097, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9431638717651367, "reward_repeat_soft_std": 0.037652309983968735, "reward_judge_quality_mean": 0.14124999940395355, "reward_judge_quality_std": 0.038335926830768585, "reward_total_composite_mean": 0.37816929817199707, "reward_total_composite_std": 0.15340575575828552} {"timestamp_utc": "2026-04-13T12:30:04Z", "mode": "train", "global_step": 2099, "epoch": 0.2108488196885987, "loss": 0.0105, "grad_norm": 15.438549041748047, "learning_rate": 3.642424242424243e-06, "num_tokens": 3775132.0, "completions/mean_length": 31.25, "completions/min_length": 27.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.25, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.44214296340942383, "rewards/meter/std": 0.4177422821521759, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9636186361312866, "rewards/repeat_soft/std": 0.0434965156018734, "rewards/judge_quality/mean": 0.1537500023841858, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.387692928314209, "rewards/total_composite/std": 0.04447413608431816, "reward": 0.387692928314209, "reward_std": 0.04447413608431816, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09119374305009842, "sampling/sampling_logp_difference/max": 1.323554515838623, "sampling/importance_sampling_ratio/min": 0.266187459230423, "sampling/importance_sampling_ratio/mean": 1.0038379430770874, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.486721184104681, "clip_ratio/low_mean": 0.06431080307811499, "clip_ratio/low_min": 0.06431080307811499, "clip_ratio/high_mean": 0.023761424235999584, "clip_ratio/high_max": 0.023761424235999584, "clip_ratio/region_mean": 0.08807222731411457, "reward_total_mean": 0.387692928314209, "reward_meter_mean": 0.44214296340942383, "reward_meter_std": 0.4177422821521759, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9636186361312866, "reward_repeat_soft_std": 0.0434965156018734, "reward_judge_quality_mean": 0.1537500023841858, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.387692928314209, "reward_total_composite_std": 0.04447413608431816} {"timestamp_utc": "2026-04-13T12:30:12Z", "mode": "train", "global_step": 2100, "epoch": 0.21094927172275238, "loss": 0.0083, "grad_norm": 6.016146659851074, "learning_rate": 3.6393939393939398e-06, "num_tokens": 3777336.0, "completions/mean_length": 102.5, "completions/min_length": 92.0, "completions/max_length": 112.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.5, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 112.0, "rewards/meter/mean": 0.8390237092971802, "rewards/meter/std": 0.1587734818458557, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.861441433429718, "rewards/repeat_soft/std": 0.029279708862304688, "rewards/judge_quality/mean": 0.18000000715255737, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4273819923400879, "rewards/total_composite/std": 0.015358088538050652, "reward": 0.4273819923400879, "reward_std": 0.015358081087470055, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07864245772361755, "sampling/sampling_logp_difference/max": 1.8556418418884277, "sampling/importance_sampling_ratio/min": 0.15635254979133606, "sampling/importance_sampling_ratio/mean": 0.9945199489593506, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4030364230275154, "clip_ratio/low_mean": 0.027041461784392595, "clip_ratio/low_min": 0.027041461784392595, "clip_ratio/high_mean": 0.055136412382125854, "clip_ratio/high_max": 0.055136412382125854, "clip_ratio/region_mean": 0.08217787416651845, "reward_total_mean": 0.4273819923400879, "reward_meter_mean": 0.8390237092971802, "reward_meter_std": 0.1587734818458557, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.861441433429718, "reward_repeat_soft_std": 0.029279708862304688, "reward_judge_quality_mean": 0.18000000715255737, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4273819923400879, "reward_total_composite_std": 0.015358088538050652} {"timestamp_utc": "2026-04-13T12:31:14Z", "mode": "eval", "global_step": 2100, "epoch": 0.21094927172275238, "eval_loss": NaN, "eval_runtime": 62.2555, "eval_samples_per_second": 1.285, "eval_steps_per_second": 0.161, "eval_num_tokens": 3777336.0, "eval_completions/mean_length": 96.5875, "eval_completions/min_length": 30.9, "eval_completions/max_length": 246.9, "eval_completions/clipped_ratio": 0.05, "eval_completions/mean_terminated_length": 74.14821548461914, "eval_completions/min_terminated_length": 30.9, "eval_completions/max_terminated_length": 132.5, "eval_rewards/meter/mean": 0.7549513936042785, "eval_rewards/meter/std": 0.2847424909472466, "eval_rewards/count_adherence/mean": 0.9800000011920929, "eval_rewards/count_adherence/std": 0.051684608310461046, "eval_rewards/hard_gate/mean": 0.95, "eval_rewards/hard_gate/std": 0.11700168251991272, "eval_rewards/repeat_soft/mean": 0.9035159707069397, "eval_rewards/repeat_soft/std": 0.07267097979784012, "eval_rewards/judge_quality/mean": 0.1567500039935112, "eval_rewards/judge_quality/std": 0.03720051711425185, "eval_rewards/total_composite/mean": 0.3925432741641998, "eval_rewards/total_composite/std": 0.07047434505075216, "eval_reward": 0.3925432741641998, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.055311382934451106, "eval_sampling/sampling_logp_difference/max": 0.9022397518157959, "eval_sampling/importance_sampling_ratio/min": 0.41649395823478697, "eval_sampling/importance_sampling_ratio/mean": 1.0110464453697205, "eval_sampling/importance_sampling_ratio/max": 1.4670676112174987, "eval_entropy": 0.5670681864023208, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.3925432741641998, "eval_reward_meter_mean": 0.7549513936042785, "eval_reward_meter_std": 0.2847424909472466, "eval_reward_count_adherence_mean": 0.9800000011920929, "eval_reward_count_adherence_std": 0.051684608310461046, "eval_reward_hard_gate_mean": 0.95, "eval_reward_hard_gate_std": 0.11700168251991272, "eval_reward_repeat_soft_mean": 0.9035159707069397, "eval_reward_repeat_soft_std": 0.07267097979784012, "eval_reward_judge_quality_mean": 0.1567500039935112, "eval_reward_judge_quality_std": 0.03720051711425185, "eval_reward_total_composite_mean": 0.3925432741641998, "eval_reward_total_composite_std": 0.07047434505075216} {"timestamp_utc": "2026-04-13T12:31:29Z", "mode": "train", "global_step": 2101, "epoch": 0.2110497237569061, "loss": -0.1714, "grad_norm": 1.856912612915039, "learning_rate": 3.6363636363636366e-06, "num_tokens": 3779100.0, "completions/mean_length": 128.5, "completions/min_length": 56.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 73.71428680419922, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 89.0, "rewards/meter/mean": 0.7168045043945312, "rewards/meter/std": 0.3875977396965027, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.2357022762298584, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.952052652835846, "rewards/repeat_soft/std": 0.033339813351631165, "rewards/judge_quality/mean": 0.14125001430511475, "rewards/judge_quality/std": 0.038335926830768585, "rewards/total_composite/mean": 0.37015044689178467, "rewards/total_composite/std": 0.15158022940158844, "reward": 0.37015044689178467, "reward_std": 0.15158022940158844, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11523141711950302, "sampling/sampling_logp_difference/max": 1.797853708267212, "sampling/importance_sampling_ratio/min": 0.1656540483236313, "sampling/importance_sampling_ratio/mean": 0.9991225600242615, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7288566678762436, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.11735270358622074, "clip_ratio/high_max": 0.11735270358622074, "clip_ratio/region_mean": 0.11735270358622074, "reward_total_mean": 0.37015044689178467, "reward_meter_mean": 0.7168045043945312, "reward_meter_std": 0.3875977396965027, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.2357022762298584, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.952052652835846, "reward_repeat_soft_std": 0.033339813351631165, "reward_judge_quality_mean": 0.14125001430511475, "reward_judge_quality_std": 0.038335926830768585, "reward_total_composite_mean": 0.37015044689178467, "reward_total_composite_std": 0.15158022940158844} {"timestamp_utc": "2026-04-13T12:31:42Z", "mode": "train", "global_step": 2102, "epoch": 0.21115017579105977, "loss": -0.1172, "grad_norm": 2.055514097213745, "learning_rate": 3.633333333333334e-06, "num_tokens": 3780782.0, "completions/mean_length": 98.25, "completions/min_length": 36.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 39.142860412597656, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.44329869747161865, "rewards/meter/std": 0.40356191992759705, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9842722415924072, "rewards/repeat_soft/std": 0.024804506450891495, "rewards/judge_quality/mean": 0.1862500011920929, "rewards/judge_quality/std": 0.11475905776023865, "rewards/total_composite/mean": 0.35340049862861633, "rewards/total_composite/std": 0.14779430627822876, "reward": 0.35340049862861633, "reward_std": 0.14779430627822876, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12127834558486938, "sampling/sampling_logp_difference/max": 1.1002140045166016, "sampling/importance_sampling_ratio/min": 0.33279985189437866, "sampling/importance_sampling_ratio/mean": 1.0372201204299927, "sampling/importance_sampling_ratio/max": 1.880364179611206, "entropy": 0.7180831506848335, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.07393320184201002, "clip_ratio/high_max": 0.07393320184201002, "clip_ratio/region_mean": 0.07393320184201002, "reward_total_mean": 0.35340049862861633, "reward_meter_mean": 0.44329869747161865, "reward_meter_std": 0.40356191992759705, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9842722415924072, "reward_repeat_soft_std": 0.024804506450891495, "reward_judge_quality_mean": 0.1862500011920929, "reward_judge_quality_std": 0.11475905776023865, "reward_total_composite_mean": 0.35340049862861633, "reward_total_composite_std": 0.14779430627822876} {"timestamp_utc": "2026-04-13T12:31:48Z", "mode": "train", "global_step": 2103, "epoch": 0.21125062782521345, "loss": 0.0086, "grad_norm": 11.103949546813965, "learning_rate": 3.6303030303030307e-06, "num_tokens": 3782484.0, "completions/mean_length": 48.75, "completions/min_length": 45.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.75, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.442565381526947, "rewards/meter/std": 0.30177053809165955, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9805001020431519, "rewards/repeat_soft/std": 0.017646679654717445, "rewards/judge_quality/mean": 0.14625000953674316, "rewards/judge_quality/std": 0.010606604628264904, "rewards/total_composite/mean": 0.39005687832832336, "rewards/total_composite/std": 0.02893988788127899, "reward": 0.39005687832832336, "reward_std": 0.02893988974392414, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1269722580909729, "sampling/sampling_logp_difference/max": 2.509066343307495, "sampling/importance_sampling_ratio/min": 0.08134415000677109, "sampling/importance_sampling_ratio/mean": 1.005758285522461, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.807975247502327, "clip_ratio/low_mean": 0.07524633966386318, "clip_ratio/low_min": 0.07524633966386318, "clip_ratio/high_mean": 0.05287114903330803, "clip_ratio/high_max": 0.05287114903330803, "clip_ratio/region_mean": 0.1281174886971712, "reward_total_mean": 0.39005687832832336, "reward_meter_mean": 0.442565381526947, "reward_meter_std": 0.30177053809165955, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9805001020431519, "reward_repeat_soft_std": 0.017646679654717445, "reward_judge_quality_mean": 0.14625000953674316, "reward_judge_quality_std": 0.010606604628264904, "reward_total_composite_mean": 0.39005687832832336, "reward_total_composite_std": 0.02893988788127899} {"timestamp_utc": "2026-04-13T12:31:55Z", "mode": "train", "global_step": 2104, "epoch": 0.21135107985936716, "loss": -0.0374, "grad_norm": 10.681594848632812, "learning_rate": 3.6272727272727275e-06, "num_tokens": 3784613.0, "completions/mean_length": 88.125, "completions/min_length": 71.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 88.125, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.6759259700775146, "rewards/meter/std": 0.307078093290329, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9187784194946289, "rewards/repeat_soft/std": 0.04656059294939041, "rewards/judge_quality/mean": 0.16875001788139343, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.4134945273399353, "rewards/total_composite/std": 0.036576032638549805, "reward": 0.4134945273399353, "reward_std": 0.0365760400891304, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11971750110387802, "sampling/sampling_logp_difference/max": 2.2448527812957764, "sampling/importance_sampling_ratio/min": 0.10594313591718674, "sampling/importance_sampling_ratio/mean": 0.9974398612976074, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.43947426602244377, "clip_ratio/low_mean": 0.03905245568603277, "clip_ratio/low_min": 0.03905245568603277, "clip_ratio/high_mean": 0.07308703940361738, "clip_ratio/high_max": 0.07308703940361738, "clip_ratio/region_mean": 0.11213949508965015, "reward_total_mean": 0.4134945273399353, "reward_meter_mean": 0.6759259700775146, "reward_meter_std": 0.307078093290329, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9187784194946289, "reward_repeat_soft_std": 0.04656059294939041, "reward_judge_quality_mean": 0.16875001788139343, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.4134945273399353, "reward_total_composite_std": 0.036576032638549805} {"timestamp_utc": "2026-04-13T12:32:01Z", "mode": "train", "global_step": 2105, "epoch": 0.21145153189352084, "loss": 0.0628, "grad_norm": 12.617268562316895, "learning_rate": 3.6242424242424248e-06, "num_tokens": 3786334.0, "completions/mean_length": 48.125, "completions/min_length": 43.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.125, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.6768026947975159, "rewards/meter/std": 0.35109129548072815, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9743654727935791, "rewards/repeat_soft/std": 0.02952939085662365, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4121430814266205, "rewards/total_composite/std": 0.034643322229385376, "reward": 0.4121430814266205, "reward_std": 0.034643322229385376, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12021853029727936, "sampling/sampling_logp_difference/max": 2.3435730934143066, "sampling/importance_sampling_ratio/min": 0.0959840640425682, "sampling/importance_sampling_ratio/mean": 1.0105738639831543, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8007223606109619, "clip_ratio/low_mean": 0.03388928063213825, "clip_ratio/low_min": 0.03388928063213825, "clip_ratio/high_mean": 0.08387564588338137, "clip_ratio/high_max": 0.08387564588338137, "clip_ratio/region_mean": 0.11776492651551962, "reward_total_mean": 0.4121430814266205, "reward_meter_mean": 0.6768026947975159, "reward_meter_std": 0.35109129548072815, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9743654727935791, "reward_repeat_soft_std": 0.02952939085662365, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4121430814266205, "reward_total_composite_std": 0.034643322229385376} {"timestamp_utc": "2026-04-13T12:32:09Z", "mode": "train", "global_step": 2106, "epoch": 0.21155198392767455, "loss": 0.0452, "grad_norm": 9.996867179870605, "learning_rate": 3.6212121212121216e-06, "num_tokens": 3788346.0, "completions/mean_length": 84.5, "completions/min_length": 73.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.5, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.6127023100852966, "rewards/meter/std": 0.33788344264030457, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.950439453125, "rewards/repeat_soft/std": 0.04181034490466118, "rewards/judge_quality/mean": 0.1612500101327896, "rewards/judge_quality/std": 0.022320717573165894, "rewards/total_composite/mean": 0.40567585825920105, "rewards/total_composite/std": 0.03350796550512314, "reward": 0.40567585825920105, "reward_std": 0.03350795805454254, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11898557841777802, "sampling/sampling_logp_difference/max": 2.1936473846435547, "sampling/importance_sampling_ratio/min": 0.1115092858672142, "sampling/importance_sampling_ratio/mean": 1.0094692707061768, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7003876119852066, "clip_ratio/low_mean": 0.0763331362977624, "clip_ratio/low_min": 0.0763331362977624, "clip_ratio/high_mean": 0.05873551592230797, "clip_ratio/high_max": 0.05873551592230797, "clip_ratio/region_mean": 0.13506865222007036, "reward_total_mean": 0.40567585825920105, "reward_meter_mean": 0.6127023100852966, "reward_meter_std": 0.33788344264030457, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.950439453125, "reward_repeat_soft_std": 0.04181034490466118, "reward_judge_quality_mean": 0.1612500101327896, "reward_judge_quality_std": 0.022320717573165894, "reward_total_composite_mean": 0.40567585825920105, "reward_total_composite_std": 0.03350796550512314} {"timestamp_utc": "2026-04-13T12:32:17Z", "mode": "train", "global_step": 2107, "epoch": 0.21165243596182823, "loss": 0.0247, "grad_norm": 12.233802795410156, "learning_rate": 3.6181818181818184e-06, "num_tokens": 3789996.0, "completions/mean_length": 41.25, "completions/min_length": 38.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.25, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.9781842231750488, "rewards/meter/std": 0.015315087512135506, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.7953553199768066, "rewards/repeat_soft/std": 0.11163556575775146, "rewards/judge_quality/mean": 0.13875000178813934, "rewards/judge_quality/std": 0.022320719435811043, "rewards/total_composite/mean": 0.40734606981277466, "rewards/total_composite/std": 0.02720000222325325, "reward": 0.40734606981277466, "reward_std": 0.027200007811188698, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09198253601789474, "sampling/sampling_logp_difference/max": 1.8582334518432617, "sampling/importance_sampling_ratio/min": 0.15594787895679474, "sampling/importance_sampling_ratio/mean": 1.0048346519470215, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3964330926537514, "clip_ratio/low_mean": 0.027148242574185133, "clip_ratio/low_min": 0.027148242574185133, "clip_ratio/high_mean": 0.03724791551940143, "clip_ratio/high_max": 0.03724791551940143, "clip_ratio/region_mean": 0.06439615809358656, "reward_total_mean": 0.40734606981277466, "reward_meter_mean": 0.9781842231750488, "reward_meter_std": 0.015315087512135506, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.7953553199768066, "reward_repeat_soft_std": 0.11163556575775146, "reward_judge_quality_mean": 0.13875000178813934, "reward_judge_quality_std": 0.022320719435811043, "reward_total_composite_mean": 0.40734606981277466, "reward_total_composite_std": 0.02720000222325325} {"timestamp_utc": "2026-04-13T12:32:23Z", "mode": "train", "global_step": 2108, "epoch": 0.2117528879959819, "loss": 0.0295, "grad_norm": 11.737648963928223, "learning_rate": 3.6151515151515153e-06, "num_tokens": 3791689.0, "completions/mean_length": 38.625, "completions/min_length": 35.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.625, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.46569669246673584, "rewards/meter/std": 0.4034918248653412, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9199074506759644, "rewards/repeat_soft/std": 0.04545706883072853, "rewards/judge_quality/mean": 0.16124999523162842, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.38604602217674255, "rewards/total_composite/std": 0.042396180331707, "reward": 0.38604602217674255, "reward_std": 0.0423961766064167, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08647025376558304, "sampling/sampling_logp_difference/max": 2.448047161102295, "sampling/importance_sampling_ratio/min": 0.08646226674318314, "sampling/importance_sampling_ratio/mean": 1.0085159540176392, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.41267629340291023, "clip_ratio/low_mean": 0.054619363974779844, "clip_ratio/low_min": 0.054619363974779844, "clip_ratio/high_mean": 0.016964286100119352, "clip_ratio/high_max": 0.016964286100119352, "clip_ratio/region_mean": 0.0715836500748992, "reward_total_mean": 0.38604602217674255, "reward_meter_mean": 0.46569669246673584, "reward_meter_std": 0.4034918248653412, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9199074506759644, "reward_repeat_soft_std": 0.04545706883072853, "reward_judge_quality_mean": 0.16124999523162842, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.38604602217674255, "reward_total_composite_std": 0.042396180331707} {"timestamp_utc": "2026-04-13T12:32:30Z", "mode": "train", "global_step": 2109, "epoch": 0.21185334003013562, "loss": 0.1557, "grad_norm": 13.237957000732422, "learning_rate": 3.6121212121212125e-06, "num_tokens": 3793245.0, "completions/mean_length": 38.5, "completions/min_length": 30.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.5, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.33292731642723083, "rewards/meter/std": 0.2673882842063904, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9947908520698547, "rewards/repeat_soft/std": 0.007985309697687626, "rewards/judge_quality/mean": 0.19500000774860382, "rewards/judge_quality/std": 0.10392305254936218, "rewards/total_composite/mean": 0.3842517137527466, "rewards/total_composite/std": 0.025468753650784492, "reward": 0.3842517137527466, "reward_std": 0.02546876110136509, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11784003674983978, "sampling/sampling_logp_difference/max": 2.1868762969970703, "sampling/importance_sampling_ratio/min": 0.11226688325405121, "sampling/importance_sampling_ratio/mean": 1.0069389343261719, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5103457905352116, "clip_ratio/low_mean": 0.05624600313603878, "clip_ratio/low_min": 0.05624600313603878, "clip_ratio/high_mean": 0.03729429328814149, "clip_ratio/high_max": 0.03729429328814149, "clip_ratio/region_mean": 0.09354029642418027, "reward_total_mean": 0.3842517137527466, "reward_meter_mean": 0.33292731642723083, "reward_meter_std": 0.2673882842063904, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9947908520698547, "reward_repeat_soft_std": 0.007985309697687626, "reward_judge_quality_mean": 0.19500000774860382, "reward_judge_quality_std": 0.10392305254936218, "reward_total_composite_mean": 0.3842517137527466, "reward_total_composite_std": 0.025468753650784492} {"timestamp_utc": "2026-04-13T12:32:37Z", "mode": "train", "global_step": 2110, "epoch": 0.2119537920642893, "loss": 0.0113, "grad_norm": 13.800068855285645, "learning_rate": 3.6090909090909093e-06, "num_tokens": 3794772.0, "completions/mean_length": 41.875, "completions/min_length": 36.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.875, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.8712973594665527, "rewards/meter/std": 0.18867942690849304, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.918583333492279, "rewards/repeat_soft/std": 0.05381178855895996, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4227389991283417, "rewards/total_composite/std": 0.022761140018701553, "reward": 0.4227389991283417, "reward_std": 0.022761141881346703, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11791300028562546, "sampling/sampling_logp_difference/max": 1.574312686920166, "sampling/importance_sampling_ratio/min": 0.20714989304542542, "sampling/importance_sampling_ratio/mean": 1.0151242017745972, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5558187812566757, "clip_ratio/low_mean": 0.02714580879546702, "clip_ratio/low_min": 0.02714580879546702, "clip_ratio/high_mean": 0.06695579458028078, "clip_ratio/high_max": 0.06695579458028078, "clip_ratio/region_mean": 0.0941016033757478, "reward_total_mean": 0.4227389991283417, "reward_meter_mean": 0.8712973594665527, "reward_meter_std": 0.18867942690849304, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.918583333492279, "reward_repeat_soft_std": 0.05381178855895996, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4227389991283417, "reward_total_composite_std": 0.022761140018701553} {"timestamp_utc": "2026-04-13T12:32:44Z", "mode": "train", "global_step": 2111, "epoch": 0.212054244098443, "loss": 0.0434, "grad_norm": 9.104913711547852, "learning_rate": 3.606060606060606e-06, "num_tokens": 3796749.0, "completions/mean_length": 89.125, "completions/min_length": 70.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.125, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.6411469578742981, "rewards/meter/std": 0.302879273891449, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.10690449178218842, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.6518181562423706, "rewards/repeat_soft/std": 0.11911094933748245, "rewards/judge_quality/mean": 0.13875000178813934, "rewards/judge_quality/std": 0.022320719435811043, "rewards/total_composite/mean": 0.33592337369918823, "rewards/total_composite/std": 0.020304610952734947, "reward": 0.33592337369918823, "reward_std": 0.020304612815380096, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07979433983564377, "sampling/sampling_logp_difference/max": 1.9750046730041504, "sampling/importance_sampling_ratio/min": 0.13876067101955414, "sampling/importance_sampling_ratio/mean": 1.0048784017562866, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.40958814322948456, "clip_ratio/low_mean": 0.0406727041117847, "clip_ratio/low_min": 0.0406727041117847, "clip_ratio/high_mean": 0.036213971907272935, "clip_ratio/high_max": 0.036213971907272935, "clip_ratio/region_mean": 0.07688667601905763, "reward_total_mean": 0.33592337369918823, "reward_meter_mean": 0.6411469578742981, "reward_meter_std": 0.302879273891449, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.10690449178218842, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.6518181562423706, "reward_repeat_soft_std": 0.11911094933748245, "reward_judge_quality_mean": 0.13875000178813934, "reward_judge_quality_std": 0.022320719435811043, "reward_total_composite_mean": 0.33592337369918823, "reward_total_composite_std": 0.020304610952734947} {"timestamp_utc": "2026-04-13T12:32:51Z", "mode": "train", "global_step": 2112, "epoch": 0.2121546961325967, "loss": -0.0532, "grad_norm": 12.300381660461426, "learning_rate": 3.603030303030303e-06, "num_tokens": 3798369.0, "completions/mean_length": 42.5, "completions/min_length": 37.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.5, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.4879136383533478, "rewards/meter/std": 0.4658832550048828, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8460896015167236, "rewards/repeat_soft/std": 0.058859921991825104, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.3744850158691406, "rewards/total_composite/std": 0.045250579714775085, "reward": 0.3744850158691406, "reward_std": 0.04525057598948479, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11138882488012314, "sampling/sampling_logp_difference/max": 1.1025335788726807, "sampling/importance_sampling_ratio/min": 0.3320287764072418, "sampling/importance_sampling_ratio/mean": 1.0110188722610474, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6444509625434875, "clip_ratio/low_mean": 0.0528691983781755, "clip_ratio/low_min": 0.0528691983781755, "clip_ratio/high_mean": 0.0698704281821847, "clip_ratio/high_max": 0.0698704281821847, "clip_ratio/region_mean": 0.1227396265603602, "reward_total_mean": 0.3744850158691406, "reward_meter_mean": 0.4879136383533478, "reward_meter_std": 0.4658832550048828, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8460896015167236, "reward_repeat_soft_std": 0.058859921991825104, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.3744850158691406, "reward_total_composite_std": 0.045250579714775085} {"timestamp_utc": "2026-04-13T12:32:57Z", "mode": "train", "global_step": 2113, "epoch": 0.21225514816675037, "loss": 0.0482, "grad_norm": 16.208444595336914, "learning_rate": 3.6000000000000003e-06, "num_tokens": 3799785.0, "completions/mean_length": 22.0, "completions/min_length": 16.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.0, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.6580546498298645, "rewards/meter/std": 0.41127723455429077, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9534668326377869, "rewards/repeat_soft/std": 0.014675324782729149, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4071803689002991, "rewards/total_composite/std": 0.038945745676755905, "reward": 0.4071803689002991, "reward_std": 0.03894574195146561, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13734646141529083, "sampling/sampling_logp_difference/max": 1.8436002731323242, "sampling/importance_sampling_ratio/min": 0.15824668109416962, "sampling/importance_sampling_ratio/mean": 0.9922757744789124, "sampling/importance_sampling_ratio/max": 1.6984217166900635, "entropy": 0.7180051393806934, "clip_ratio/low_mean": 0.047727273777127266, "clip_ratio/low_min": 0.047727273777127266, "clip_ratio/high_mean": 0.09421969391405582, "clip_ratio/high_max": 0.09421969391405582, "clip_ratio/region_mean": 0.1419469676911831, "reward_total_mean": 0.4071803689002991, "reward_meter_mean": 0.6580546498298645, "reward_meter_std": 0.41127723455429077, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9534668326377869, "reward_repeat_soft_std": 0.014675324782729149, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4071803689002991, "reward_total_composite_std": 0.038945745676755905} {"timestamp_utc": "2026-04-13T12:33:08Z", "mode": "train", "global_step": 2114, "epoch": 0.21235560020090408, "loss": -0.1491, "grad_norm": 1.8867636919021606, "learning_rate": 3.596969696969697e-06, "num_tokens": 3801567.0, "completions/mean_length": 115.75, "completions/min_length": 49.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 59.142860412597656, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.5922025442123413, "rewards/meter/std": 0.3228558599948883, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.971898078918457, "rewards/repeat_soft/std": 0.02474227547645569, "rewards/judge_quality/mean": 0.1525000035762787, "rewards/judge_quality/std": 0.04399675503373146, "rewards/total_composite/mean": 0.36274904012680054, "rewards/total_composite/std": 0.1502305567264557, "reward": 0.36274904012680054, "reward_std": 0.1502305418252945, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11633001267910004, "sampling/sampling_logp_difference/max": 3.0924201011657715, "sampling/importance_sampling_ratio/min": 0.04539196565747261, "sampling/importance_sampling_ratio/mean": 0.9857916235923767, "sampling/importance_sampling_ratio/max": 1.9030864238739014, "entropy": 0.47304895520210266, "clip_ratio/low_mean": 0.01024590153247118, "clip_ratio/low_min": 0.01024590153247118, "clip_ratio/high_mean": 0.08157390914857388, "clip_ratio/high_max": 0.08157390914857388, "clip_ratio/region_mean": 0.09181981068104506, "reward_total_mean": 0.36274904012680054, "reward_meter_mean": 0.5922025442123413, "reward_meter_std": 0.3228558599948883, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.971898078918457, "reward_repeat_soft_std": 0.02474227547645569, "reward_judge_quality_mean": 0.1525000035762787, "reward_judge_quality_std": 0.04399675503373146, "reward_total_composite_mean": 0.36274904012680054, "reward_total_composite_std": 0.1502305567264557} {"timestamp_utc": "2026-04-13T12:33:14Z", "mode": "train", "global_step": 2115, "epoch": 0.21245605223505776, "loss": 0.0127, "grad_norm": 11.018208503723145, "learning_rate": 3.593939393939394e-06, "num_tokens": 3803102.0, "completions/mean_length": 43.875, "completions/min_length": 38.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.875, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.8543543219566345, "rewards/meter/std": 0.28948941826820374, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9505391120910645, "rewards/repeat_soft/std": 0.01934647373855114, "rewards/judge_quality/mean": 0.1537500023841858, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.4282057285308838, "rewards/total_composite/std": 0.03167116641998291, "reward": 0.4282057285308838, "reward_std": 0.03167115896940231, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10174911469221115, "sampling/sampling_logp_difference/max": 1.6101007461547852, "sampling/importance_sampling_ratio/min": 0.19986747205257416, "sampling/importance_sampling_ratio/mean": 0.9898253679275513, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4199068732559681, "clip_ratio/low_mean": 0.011627906933426857, "clip_ratio/low_min": 0.011627906933426857, "clip_ratio/high_mean": 0.1053868168964982, "clip_ratio/high_max": 0.1053868168964982, "clip_ratio/region_mean": 0.11701472382992506, "reward_total_mean": 0.4282057285308838, "reward_meter_mean": 0.8543543219566345, "reward_meter_std": 0.28948941826820374, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9505391120910645, "reward_repeat_soft_std": 0.01934647373855114, "reward_judge_quality_mean": 0.1537500023841858, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.4282057285308838, "reward_total_composite_std": 0.03167116641998291} {"timestamp_utc": "2026-04-13T12:33:21Z", "mode": "train", "global_step": 2116, "epoch": 0.21255650426921144, "loss": 0.0131, "grad_norm": 9.96137809753418, "learning_rate": 3.590909090909091e-06, "num_tokens": 3804769.0, "completions/mean_length": 51.375, "completions/min_length": 45.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.375, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.610297441482544, "rewards/meter/std": 0.37020719051361084, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9516139030456543, "rewards/repeat_soft/std": 0.03949163854122162, "rewards/judge_quality/mean": 0.25, "rewards/judge_quality/std": 0.2709243595600128, "rewards/total_composite/mean": 0.46495306491851807, "rewards/total_composite/std": 0.1915210336446762, "reward": 0.46495306491851807, "reward_std": 0.1915210336446762, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0902131050825119, "sampling/sampling_logp_difference/max": 1.5951628684997559, "sampling/importance_sampling_ratio/min": 0.20287548005580902, "sampling/importance_sampling_ratio/mean": 1.0204806327819824, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5200207754969597, "clip_ratio/low_mean": 0.08550144638866186, "clip_ratio/low_min": 0.08550144638866186, "clip_ratio/high_mean": 0.009433962404727936, "clip_ratio/high_max": 0.009433962404727936, "clip_ratio/region_mean": 0.0949354087933898, "reward_total_mean": 0.46495306491851807, "reward_meter_mean": 0.610297441482544, "reward_meter_std": 0.37020719051361084, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9516139030456543, "reward_repeat_soft_std": 0.03949163854122162, "reward_judge_quality_mean": 0.25, "reward_judge_quality_std": 0.2709243595600128, "reward_total_composite_mean": 0.46495306491851807, "reward_total_composite_std": 0.1915210336446762} {"timestamp_utc": "2026-04-13T12:33:27Z", "mode": "train", "global_step": 2117, "epoch": 0.21265695630336515, "loss": -0.0345, "grad_norm": 10.987751960754395, "learning_rate": 3.587878787878788e-06, "num_tokens": 3806689.0, "completions/mean_length": 56.0, "completions/min_length": 50.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.0, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.48459476232528687, "rewards/meter/std": 0.3567312955856323, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9572969079017639, "rewards/repeat_soft/std": 0.03782852366566658, "rewards/judge_quality/mean": 0.14625000953674316, "rewards/judge_quality/std": 0.010606604628264904, "rewards/total_composite/mean": 0.388683557510376, "rewards/total_composite/std": 0.03286159411072731, "reward": 0.388683557510376, "reward_std": 0.03286158666014671, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11895419657230377, "sampling/sampling_logp_difference/max": 3.274886131286621, "sampling/importance_sampling_ratio/min": 0.037821173667907715, "sampling/importance_sampling_ratio/mean": 1.0060267448425293, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6553157195448875, "clip_ratio/low_mean": 0.05163145810365677, "clip_ratio/low_min": 0.05163145810365677, "clip_ratio/high_mean": 0.04374784370884299, "clip_ratio/high_max": 0.04374784370884299, "clip_ratio/region_mean": 0.09537930181249976, "reward_total_mean": 0.388683557510376, "reward_meter_mean": 0.48459476232528687, "reward_meter_std": 0.3567312955856323, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9572969079017639, "reward_repeat_soft_std": 0.03782852366566658, "reward_judge_quality_mean": 0.14625000953674316, "reward_judge_quality_std": 0.010606604628264904, "reward_total_composite_mean": 0.388683557510376, "reward_total_composite_std": 0.03286159411072731} {"timestamp_utc": "2026-04-13T12:33:40Z", "mode": "train", "global_step": 2118, "epoch": 0.21275740833751883, "loss": -0.1357, "grad_norm": 1.8776116371154785, "learning_rate": 3.584848484848485e-06, "num_tokens": 3808480.0, "completions/mean_length": 106.875, "completions/min_length": 39.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 49.000003814697266, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9100183248519897, "rewards/meter/std": 0.22264239192008972, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9692996740341187, "rewards/repeat_soft/std": 0.016032908111810684, "rewards/judge_quality/mean": 0.13375000655651093, "rewards/judge_quality/std": 0.03543102368712425, "rewards/total_composite/mean": 0.384857714176178, "rewards/total_composite/std": 0.15563072264194489, "reward": 0.384857714176178, "reward_std": 0.15563073754310608, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13287843763828278, "sampling/sampling_logp_difference/max": 4.305706977844238, "sampling/importance_sampling_ratio/min": 0.013491343706846237, "sampling/importance_sampling_ratio/mean": 0.9984962344169617, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5105771720409393, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.11489598639309406, "clip_ratio/high_max": 0.11489598639309406, "clip_ratio/region_mean": 0.11489598639309406, "reward_total_mean": 0.384857714176178, "reward_meter_mean": 0.9100183248519897, "reward_meter_std": 0.22264239192008972, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9692996740341187, "reward_repeat_soft_std": 0.016032908111810684, "reward_judge_quality_mean": 0.13375000655651093, "reward_judge_quality_std": 0.03543102368712425, "reward_total_composite_mean": 0.384857714176178, "reward_total_composite_std": 0.15563072264194489} {"timestamp_utc": "2026-04-13T12:33:47Z", "mode": "train", "global_step": 2119, "epoch": 0.21285786037167254, "loss": 0.0006, "grad_norm": 5.920814514160156, "learning_rate": 3.5818181818181817e-06, "num_tokens": 3810570.0, "completions/mean_length": 85.25, "completions/min_length": 80.0, "completions/max_length": 92.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.25, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 92.0, "rewards/meter/mean": 0.9678657054901123, "rewards/meter/std": 0.02926114946603775, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8394684195518494, "rewards/repeat_soft/std": 0.03335115313529968, "rewards/judge_quality/mean": 0.18000000715255737, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.43916055560112, "rewards/total_composite/std": 0.004552391357719898, "reward": 0.43916055560112, "reward_std": 0.0045523978769779205, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07062681764364243, "sampling/sampling_logp_difference/max": 1.2569303512573242, "sampling/importance_sampling_ratio/min": 0.2845260798931122, "sampling/importance_sampling_ratio/mean": 1.002211093902588, "sampling/importance_sampling_ratio/max": 1.83241605758667, "entropy": 0.3776538223028183, "clip_ratio/low_mean": 0.03219923097640276, "clip_ratio/low_min": 0.03219923097640276, "clip_ratio/high_mean": 0.04883924592286348, "clip_ratio/high_max": 0.04883924592286348, "clip_ratio/region_mean": 0.08103847689926624, "reward_total_mean": 0.43916055560112, "reward_meter_mean": 0.9678657054901123, "reward_meter_std": 0.02926114946603775, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8394684195518494, "reward_repeat_soft_std": 0.03335115313529968, "reward_judge_quality_mean": 0.18000000715255737, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.43916055560112, "reward_total_composite_std": 0.004552391357719898} {"timestamp_utc": "2026-04-13T12:33:54Z", "mode": "train", "global_step": 2120, "epoch": 0.21295831240582622, "loss": 0.0594, "grad_norm": 18.88593101501465, "learning_rate": 3.578787878787879e-06, "num_tokens": 3812212.0, "completions/mean_length": 37.25, "completions/min_length": 33.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.25, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.8623172044754028, "rewards/meter/std": 0.2601478397846222, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9530335664749146, "rewards/repeat_soft/std": 0.05007192865014076, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4270309805870056, "rewards/total_composite/std": 0.02340822108089924, "reward": 0.4270309805870056, "reward_std": 0.023408222943544388, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09455607831478119, "sampling/sampling_logp_difference/max": 2.033214807510376, "sampling/importance_sampling_ratio/min": 0.13091398775577545, "sampling/importance_sampling_ratio/mean": 1.0077043771743774, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.375054020434618, "clip_ratio/low_mean": 0.019736841320991516, "clip_ratio/low_min": 0.019736841320991516, "clip_ratio/high_mean": 0.0718057572375983, "clip_ratio/high_max": 0.0718057572375983, "clip_ratio/region_mean": 0.09154259855858982, "reward_total_mean": 0.4270309805870056, "reward_meter_mean": 0.8623172044754028, "reward_meter_std": 0.2601478397846222, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9530335664749146, "reward_repeat_soft_std": 0.05007192865014076, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4270309805870056, "reward_total_composite_std": 0.02340822108089924} {"timestamp_utc": "2026-04-13T12:34:00Z", "mode": "train", "global_step": 2121, "epoch": 0.2130587644399799, "loss": -0.1005, "grad_norm": 16.1114444732666, "learning_rate": 3.575757575757576e-06, "num_tokens": 3813721.0, "completions/mean_length": 27.625, "completions/min_length": 24.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.625, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.34160304069519043, "rewards/meter/std": 0.3914759159088135, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9752987623214722, "rewards/repeat_soft/std": 0.004002916160970926, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.3796011209487915, "rewards/total_composite/std": 0.03823774307966232, "reward": 0.3796011209487915, "reward_std": 0.03823775053024292, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08197850733995438, "sampling/sampling_logp_difference/max": 2.0052106380462646, "sampling/importance_sampling_ratio/min": 0.1346319317817688, "sampling/importance_sampling_ratio/mean": 0.9941287636756897, "sampling/importance_sampling_ratio/max": 1.8110864162445068, "entropy": 0.3139124680310488, "clip_ratio/low_mean": 0.03363671526312828, "clip_ratio/low_min": 0.03363671526312828, "clip_ratio/high_mean": 0.02470588218420744, "clip_ratio/high_max": 0.02470588218420744, "clip_ratio/region_mean": 0.05834259744733572, "reward_total_mean": 0.3796011209487915, "reward_meter_mean": 0.34160304069519043, "reward_meter_std": 0.3914759159088135, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9752987623214722, "reward_repeat_soft_std": 0.004002916160970926, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.3796011209487915, "reward_total_composite_std": 0.03823774307966232} {"timestamp_utc": "2026-04-13T12:34:07Z", "mode": "train", "global_step": 2122, "epoch": 0.2131592164741336, "loss": -0.0917, "grad_norm": 20.8836727142334, "learning_rate": 3.5727272727272734e-06, "num_tokens": 3815112.0, "completions/mean_length": 22.875, "completions/min_length": 16.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.875, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.6766870021820068, "rewards/meter/std": 0.4345723092556, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9576764702796936, "rewards/repeat_soft/std": 0.013642913661897182, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4096284508705139, "rewards/total_composite/std": 0.04190043732523918, "reward": 0.4096284508705139, "reward_std": 0.04190043359994888, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16140659153461456, "sampling/sampling_logp_difference/max": 1.4260950088500977, "sampling/importance_sampling_ratio/min": 0.2402452528476715, "sampling/importance_sampling_ratio/mean": 0.9993340969085693, "sampling/importance_sampling_ratio/max": 1.772874116897583, "entropy": 0.9253654107451439, "clip_ratio/low_mean": 0.06414473708719015, "clip_ratio/low_min": 0.06414473708719015, "clip_ratio/high_mean": 0.12481671385467052, "clip_ratio/high_max": 0.12481671385467052, "clip_ratio/region_mean": 0.18896145094186068, "reward_total_mean": 0.4096284508705139, "reward_meter_mean": 0.6766870021820068, "reward_meter_std": 0.4345723092556, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9576764702796936, "reward_repeat_soft_std": 0.013642913661897182, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4096284508705139, "reward_total_composite_std": 0.04190043732523918} {"timestamp_utc": "2026-04-13T12:34:14Z", "mode": "train", "global_step": 2123, "epoch": 0.2132596685082873, "loss": 0.0178, "grad_norm": 12.174962043762207, "learning_rate": 3.5696969696969703e-06, "num_tokens": 3816764.0, "completions/mean_length": 47.5, "completions/min_length": 40.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.5, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.8043650984764099, "rewards/meter/std": 0.3080444037914276, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9377086758613586, "rewards/repeat_soft/std": 0.02444099448621273, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4190818667411804, "rewards/total_composite/std": 0.029328929260373116, "reward": 0.4190818667411804, "reward_std": 0.02932891622185707, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08279678970575333, "sampling/sampling_logp_difference/max": 1.8162415027618408, "sampling/importance_sampling_ratio/min": 0.16263586282730103, "sampling/importance_sampling_ratio/mean": 0.9955849647521973, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4308505542576313, "clip_ratio/low_mean": 0.0252380957826972, "clip_ratio/low_min": 0.0252380957826972, "clip_ratio/high_mean": 0.046594975516200066, "clip_ratio/high_max": 0.046594975516200066, "clip_ratio/region_mean": 0.07183307129889727, "reward_total_mean": 0.4190818667411804, "reward_meter_mean": 0.8043650984764099, "reward_meter_std": 0.3080444037914276, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9377086758613586, "reward_repeat_soft_std": 0.02444099448621273, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4190818667411804, "reward_total_composite_std": 0.029328929260373116} {"timestamp_utc": "2026-04-13T12:34:21Z", "mode": "train", "global_step": 2124, "epoch": 0.213360120542441, "loss": 0.0328, "grad_norm": 9.952363014221191, "learning_rate": 3.566666666666667e-06, "num_tokens": 3818445.0, "completions/mean_length": 51.125, "completions/min_length": 45.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.125, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.9706482887268066, "rewards/meter/std": 0.039042141288518906, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.877032995223999, "rewards/repeat_soft/std": 0.07919418066740036, "rewards/judge_quality/mean": 0.14249999821186066, "rewards/judge_quality/std": 0.02121320739388466, "rewards/total_composite/mean": 0.4213140606880188, "rewards/total_composite/std": 0.02052641287446022, "reward": 0.4213140606880188, "reward_std": 0.020526403561234474, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07824968546628952, "sampling/sampling_logp_difference/max": 1.197680950164795, "sampling/importance_sampling_ratio/min": 0.3058168292045593, "sampling/importance_sampling_ratio/mean": 0.9943525791168213, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2968391440808773, "clip_ratio/low_mean": 0.02626256551593542, "clip_ratio/low_min": 0.02626256551593542, "clip_ratio/high_mean": 0.030169055331498384, "clip_ratio/high_max": 0.030169055331498384, "clip_ratio/region_mean": 0.056431620847433805, "reward_total_mean": 0.4213140606880188, "reward_meter_mean": 0.9706482887268066, "reward_meter_std": 0.039042141288518906, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.877032995223999, "reward_repeat_soft_std": 0.07919418066740036, "reward_judge_quality_mean": 0.14249999821186066, "reward_judge_quality_std": 0.02121320739388466, "reward_total_composite_mean": 0.4213140606880188, "reward_total_composite_std": 0.02052641287446022} {"timestamp_utc": "2026-04-13T12:34:29Z", "mode": "train", "global_step": 2125, "epoch": 0.21346057257659468, "loss": -0.0465, "grad_norm": 7.96971321105957, "learning_rate": 3.563636363636364e-06, "num_tokens": 3820811.0, "completions/mean_length": 115.75, "completions/min_length": 96.0, "completions/max_length": 128.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.75, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 128.0, "rewards/meter/mean": 0.8100929260253906, "rewards/meter/std": 0.3253575265407562, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8903893232345581, "rewards/repeat_soft/std": 0.017612675204873085, "rewards/judge_quality/mean": 0.1237500011920929, "rewards/judge_quality/std": 0.010606604628264904, "rewards/total_composite/mean": 0.3969489634037018, "rewards/total_composite/std": 0.023812510073184967, "reward": 0.3969489634037018, "reward_std": 0.023812511935830116, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09560476988554001, "sampling/sampling_logp_difference/max": 1.885268211364746, "sampling/importance_sampling_ratio/min": 0.1517883539199829, "sampling/importance_sampling_ratio/mean": 1.0107030868530273, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.46189267933368683, "clip_ratio/low_mean": 0.022395833395421505, "clip_ratio/low_min": 0.022395833395421505, "clip_ratio/high_mean": 0.07545540388673544, "clip_ratio/high_max": 0.07545540388673544, "clip_ratio/region_mean": 0.09785123728215694, "reward_total_mean": 0.3969489634037018, "reward_meter_mean": 0.8100929260253906, "reward_meter_std": 0.3253575265407562, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8903893232345581, "reward_repeat_soft_std": 0.017612675204873085, "reward_judge_quality_mean": 0.1237500011920929, "reward_judge_quality_std": 0.010606604628264904, "reward_total_composite_mean": 0.3969489634037018, "reward_total_composite_std": 0.023812510073184967} {"timestamp_utc": "2026-04-13T12:34:40Z", "mode": "train", "global_step": 2126, "epoch": 0.21356102461074836, "loss": -0.1275, "grad_norm": 2.2829623222351074, "learning_rate": 3.560606060606061e-06, "num_tokens": 3822445.0, "completions/mean_length": 110.25, "completions/min_length": 48.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 52.85714340209961, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.7273818254470825, "rewards/meter/std": 0.42327576875686646, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9806386232376099, "rewards/repeat_soft/std": 0.020722202956676483, "rewards/judge_quality/mean": 0.17875000834465027, "rewards/judge_quality/std": 0.11605878919363022, "rewards/total_composite/mean": 0.39845943450927734, "rewards/total_composite/std": 0.17626473307609558, "reward": 0.39845943450927734, "reward_std": 0.1762647181749344, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10056637227535248, "sampling/sampling_logp_difference/max": 3.5167078971862793, "sampling/importance_sampling_ratio/min": 0.029697038233280182, "sampling/importance_sampling_ratio/mean": 1.0091761350631714, "sampling/importance_sampling_ratio/max": 1.9448227882385254, "entropy": 0.5782650634646416, "clip_ratio/low_mean": 0.012500000186264515, "clip_ratio/low_min": 0.012500000186264515, "clip_ratio/high_mean": 0.07084880443289876, "clip_ratio/high_max": 0.07084880443289876, "clip_ratio/region_mean": 0.08334880461916327, "reward_total_mean": 0.39845943450927734, "reward_meter_mean": 0.7273818254470825, "reward_meter_std": 0.42327576875686646, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9806386232376099, "reward_repeat_soft_std": 0.020722202956676483, "reward_judge_quality_mean": 0.17875000834465027, "reward_judge_quality_std": 0.11605878919363022, "reward_total_composite_mean": 0.39845943450927734, "reward_total_composite_std": 0.17626473307609558} {"timestamp_utc": "2026-04-13T12:34:47Z", "mode": "train", "global_step": 2127, "epoch": 0.21366147664490207, "loss": 0.0744, "grad_norm": 10.888141632080078, "learning_rate": 3.557575757575758e-06, "num_tokens": 3824123.0, "completions/mean_length": 41.75, "completions/min_length": 33.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.75, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.961769700050354, "rewards/meter/std": 0.04579383134841919, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9323242902755737, "rewards/repeat_soft/std": 0.029701024293899536, "rewards/judge_quality/mean": 0.22500000894069672, "rewards/judge_quality/std": 0.13887301087379456, "rewards/total_composite/mean": 0.48145878314971924, "rewards/total_composite/std": 0.08942937850952148, "reward": 0.48145878314971924, "reward_std": 0.08942936360836029, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08970857411623001, "sampling/sampling_logp_difference/max": 1.39491605758667, "sampling/importance_sampling_ratio/min": 0.2478538304567337, "sampling/importance_sampling_ratio/mean": 0.9984186887741089, "sampling/importance_sampling_ratio/max": 1.674881935119629, "entropy": 0.5424296446144581, "clip_ratio/low_mean": 0.03404927044175565, "clip_ratio/low_min": 0.03404927044175565, "clip_ratio/high_mean": 0.029656318947672844, "clip_ratio/high_max": 0.029656318947672844, "clip_ratio/region_mean": 0.0637055893894285, "reward_total_mean": 0.48145878314971924, "reward_meter_mean": 0.961769700050354, "reward_meter_std": 0.04579383134841919, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9323242902755737, "reward_repeat_soft_std": 0.029701024293899536, "reward_judge_quality_mean": 0.22500000894069672, "reward_judge_quality_std": 0.13887301087379456, "reward_total_composite_mean": 0.48145878314971924, "reward_total_composite_std": 0.08942937850952148} {"timestamp_utc": "2026-04-13T12:35:01Z", "mode": "train", "global_step": 2128, "epoch": 0.21376192867905575, "loss": -0.1926, "grad_norm": 1.7553354501724243, "learning_rate": 3.554545454545455e-06, "num_tokens": 3826618.0, "completions/mean_length": 161.875, "completions/min_length": 103.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 111.85714721679688, "completions/min_terminated_length": 103.0, "completions/max_terminated_length": 130.0, "rewards/meter/mean": 0.8127691745758057, "rewards/meter/std": 0.2406369149684906, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8884314298629761, "rewards/repeat_soft/std": 0.041326019912958145, "rewards/judge_quality/mean": 0.14875000715255737, "rewards/judge_quality/std": 0.048236772418022156, "rewards/total_composite/mean": 0.3597673177719116, "rewards/total_composite/std": 0.15010634064674377, "reward": 0.3597673177719116, "reward_std": 0.15010634064674377, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09040025621652603, "sampling/sampling_logp_difference/max": 1.4311444759368896, "sampling/importance_sampling_ratio/min": 0.2390352040529251, "sampling/importance_sampling_ratio/mean": 1.004586100578308, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4321543797850609, "clip_ratio/low_mean": 0.014423076994717121, "clip_ratio/low_min": 0.014423076994717121, "clip_ratio/high_mean": 0.07017199601978064, "clip_ratio/high_max": 0.07017199601978064, "clip_ratio/region_mean": 0.08459507301449776, "reward_total_mean": 0.3597673177719116, "reward_meter_mean": 0.8127691745758057, "reward_meter_std": 0.2406369149684906, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8884314298629761, "reward_repeat_soft_std": 0.041326019912958145, "reward_judge_quality_mean": 0.14875000715255737, "reward_judge_quality_std": 0.048236772418022156, "reward_total_composite_mean": 0.3597673177719116, "reward_total_composite_std": 0.15010634064674377} {"timestamp_utc": "2026-04-13T12:35:08Z", "mode": "train", "global_step": 2129, "epoch": 0.21386238071320945, "loss": 0.065, "grad_norm": 6.841037273406982, "learning_rate": 3.551515151515152e-06, "num_tokens": 3828796.0, "completions/mean_length": 98.25, "completions/min_length": 88.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.25, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.9749183654785156, "rewards/meter/std": 0.01414470374584198, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8836276531219482, "rewards/repeat_soft/std": 0.0258223544806242, "rewards/judge_quality/mean": 0.17625001072883606, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.44421523809432983, "rewards/total_composite/std": 0.0071704210713505745, "reward": 0.44421523809432983, "reward_std": 0.007170422002673149, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09535162150859833, "sampling/sampling_logp_difference/max": 3.7799437046051025, "sampling/importance_sampling_ratio/min": 0.02282397635281086, "sampling/importance_sampling_ratio/mean": 0.9967421293258667, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3356841690838337, "clip_ratio/low_mean": 0.02442577015608549, "clip_ratio/low_min": 0.02442577015608549, "clip_ratio/high_mean": 0.04454277083277702, "clip_ratio/high_max": 0.04454277083277702, "clip_ratio/region_mean": 0.06896854098886251, "reward_total_mean": 0.44421523809432983, "reward_meter_mean": 0.9749183654785156, "reward_meter_std": 0.01414470374584198, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8836276531219482, "reward_repeat_soft_std": 0.0258223544806242, "reward_judge_quality_mean": 0.17625001072883606, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.44421523809432983, "reward_total_composite_std": 0.0071704210713505745} {"timestamp_utc": "2026-04-13T12:35:15Z", "mode": "train", "global_step": 2130, "epoch": 0.21396283274736314, "loss": -0.0544, "grad_norm": 7.538710594177246, "learning_rate": 3.548484848484849e-06, "num_tokens": 3831009.0, "completions/mean_length": 102.625, "completions/min_length": 88.0, "completions/max_length": 115.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.625, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 115.0, "rewards/meter/mean": 0.9733519554138184, "rewards/meter/std": 0.02541358955204487, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.740585446357727, "rewards/repeat_soft/std": 0.12306283414363861, "rewards/judge_quality/mean": 0.1274999976158142, "rewards/judge_quality/std": 0.02121320739388466, "rewards/total_composite/mean": 0.39185476303100586, "rewards/total_composite/std": 0.0253920815885067, "reward": 0.39185476303100586, "reward_std": 0.02539207972586155, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07328380644321442, "sampling/sampling_logp_difference/max": 1.7725296020507812, "sampling/importance_sampling_ratio/min": 0.16990265250205994, "sampling/importance_sampling_ratio/mean": 0.9985268712043762, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3542307838797569, "clip_ratio/low_mean": 0.02232967270538211, "clip_ratio/low_min": 0.02232967270538211, "clip_ratio/high_mean": 0.040749015752226114, "clip_ratio/high_max": 0.040749015752226114, "clip_ratio/region_mean": 0.06307868845760822, "reward_total_mean": 0.39185476303100586, "reward_meter_mean": 0.9733519554138184, "reward_meter_std": 0.02541358955204487, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.740585446357727, "reward_repeat_soft_std": 0.12306283414363861, "reward_judge_quality_mean": 0.1274999976158142, "reward_judge_quality_std": 0.02121320739388466, "reward_total_composite_mean": 0.39185476303100586, "reward_total_composite_std": 0.0253920815885067} {"timestamp_utc": "2026-04-13T12:35:23Z", "mode": "train", "global_step": 2131, "epoch": 0.21406328478151682, "loss": -0.0066, "grad_norm": 6.729617595672607, "learning_rate": 3.5454545454545458e-06, "num_tokens": 3833642.0, "completions/mean_length": 143.125, "completions/min_length": 133.0, "completions/max_length": 153.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 143.125, "completions/min_terminated_length": 133.0, "completions/max_terminated_length": 153.0, "rewards/meter/mean": 0.9710224270820618, "rewards/meter/std": 0.024256430566310883, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8977535963058472, "rewards/repeat_soft/std": 0.03404621779918671, "rewards/judge_quality/mean": 0.1537500023841858, "rewards/judge_quality/std": 0.029730942100286484, "rewards/total_composite/mean": 0.42175352573394775, "rewards/total_composite/std": 0.02008803002536297, "reward": 0.42175352573394775, "reward_std": 0.020088035613298416, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11049627512693405, "sampling/sampling_logp_difference/max": 2.2645256519317627, "sampling/importance_sampling_ratio/min": 0.10387929528951645, "sampling/importance_sampling_ratio/mean": 0.9980050325393677, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5583962202072144, "clip_ratio/low_mean": 0.05468147713690996, "clip_ratio/low_min": 0.05468147713690996, "clip_ratio/high_mean": 0.04470324795693159, "clip_ratio/high_max": 0.04470324795693159, "clip_ratio/region_mean": 0.09938472509384155, "reward_total_mean": 0.42175352573394775, "reward_meter_mean": 0.9710224270820618, "reward_meter_std": 0.024256430566310883, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8977535963058472, "reward_repeat_soft_std": 0.03404621779918671, "reward_judge_quality_mean": 0.1537500023841858, "reward_judge_quality_std": 0.029730942100286484, "reward_total_composite_mean": 0.42175352573394775, "reward_total_composite_std": 0.02008803002536297} {"timestamp_utc": "2026-04-13T12:35:29Z", "mode": "train", "global_step": 2132, "epoch": 0.21416373681567052, "loss": -0.04, "grad_norm": 19.79250717163086, "learning_rate": 3.5424242424242426e-06, "num_tokens": 3835091.0, "completions/mean_length": 25.125, "completions/min_length": 20.0, "completions/max_length": 28.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.125, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 28.0, "rewards/meter/mean": 0.9905524253845215, "rewards/meter/std": 0.008373869583010674, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9346094131469727, "rewards/repeat_soft/std": 0.04883209988474846, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4367702603340149, "rewards/total_composite/std": 0.007294591516256332, "reward": 0.4367702603340149, "reward_std": 0.007294599432498217, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08623015135526657, "sampling/sampling_logp_difference/max": 1.2098863124847412, "sampling/importance_sampling_ratio/min": 0.3049562871456146, "sampling/importance_sampling_ratio/mean": 1.0061790943145752, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6499927863478661, "clip_ratio/low_mean": 0.02175925951451063, "clip_ratio/low_min": 0.02175925951451063, "clip_ratio/high_mean": 0.060073030181229115, "clip_ratio/high_max": 0.060073030181229115, "clip_ratio/region_mean": 0.08183228969573975, "reward_total_mean": 0.4367702603340149, "reward_meter_mean": 0.9905524253845215, "reward_meter_std": 0.008373869583010674, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9346094131469727, "reward_repeat_soft_std": 0.04883209988474846, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4367702603340149, "reward_total_composite_std": 0.007294591516256332} {"timestamp_utc": "2026-04-13T12:35:36Z", "mode": "train", "global_step": 2133, "epoch": 0.2142641888498242, "loss": 0.0394, "grad_norm": 10.58881950378418, "learning_rate": 3.53939393939394e-06, "num_tokens": 3836604.0, "completions/mean_length": 23.125, "completions/min_length": 21.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.125, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.8037121295928955, "rewards/meter/std": 0.3496001064777374, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4227369427680969, "rewards/total_composite/std": 0.03408600762486458, "reward": 0.4227369427680969, "reward_std": 0.03408601135015488, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09222669899463654, "sampling/sampling_logp_difference/max": 1.510305404663086, "sampling/importance_sampling_ratio/min": 0.22084252536296844, "sampling/importance_sampling_ratio/mean": 1.0104715824127197, "sampling/importance_sampling_ratio/max": 1.919601321220398, "entropy": 0.3871112950146198, "clip_ratio/low_mean": 0.02003205195069313, "clip_ratio/low_min": 0.02003205195069313, "clip_ratio/high_mean": 0.044679089449346066, "clip_ratio/high_max": 0.044679089449346066, "clip_ratio/region_mean": 0.0647111414000392, "reward_total_mean": 0.4227369427680969, "reward_meter_mean": 0.8037121295928955, "reward_meter_std": 0.3496001064777374, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4227369427680969, "reward_total_composite_std": 0.03408600762486458} {"timestamp_utc": "2026-04-13T12:35:48Z", "mode": "train", "global_step": 2134, "epoch": 0.2143646408839779, "loss": -0.2088, "grad_norm": 1.5877206325531006, "learning_rate": 3.5363636363636367e-06, "num_tokens": 3838974.0, "completions/mean_length": 163.25, "completions/min_length": 100.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 113.42857360839844, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 127.0, "rewards/meter/mean": 0.9713046550750732, "rewards/meter/std": 0.03033020719885826, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.2651650309562683, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9271901845932007, "rewards/repeat_soft/std": 0.020566994324326515, "rewards/judge_quality/mean": 0.1600000113248825, "rewards/judge_quality/std": 0.04566962271928787, "rewards/total_composite/mean": 0.39359205961227417, "rewards/total_composite/std": 0.159275084733963, "reward": 0.39359205961227417, "reward_std": 0.159275084733963, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08693575859069824, "sampling/sampling_logp_difference/max": 2.0825905799865723, "sampling/importance_sampling_ratio/min": 0.12460698932409286, "sampling/importance_sampling_ratio/mean": 0.9960200786590576, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4335934594273567, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.07275812653824687, "clip_ratio/high_max": 0.07275812653824687, "clip_ratio/region_mean": 0.07275812653824687, "reward_total_mean": 0.39359205961227417, "reward_meter_mean": 0.9713046550750732, "reward_meter_std": 0.03033020719885826, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.2651650309562683, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9271901845932007, "reward_repeat_soft_std": 0.020566994324326515, "reward_judge_quality_mean": 0.1600000113248825, "reward_judge_quality_std": 0.04566962271928787, "reward_total_composite_mean": 0.39359205961227417, "reward_total_composite_std": 0.159275084733963} {"timestamp_utc": "2026-04-13T12:36:01Z", "mode": "train", "global_step": 2135, "epoch": 0.2144650929181316, "loss": -0.1097, "grad_norm": 2.3888208866119385, "learning_rate": 3.5333333333333335e-06, "num_tokens": 3840585.0, "completions/mean_length": 95.375, "completions/min_length": 33.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 35.85714340209961, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.7480518817901611, "rewards/meter/std": 0.29896217584609985, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9610674381256104, "rewards/repeat_soft/std": 0.01786000281572342, "rewards/judge_quality/mean": 0.17500001192092896, "rewards/judge_quality/std": 0.11649646610021591, "rewards/total_composite/mean": 0.39515113830566406, "rewards/total_composite/std": 0.17731983959674835, "reward": 0.39515113830566406, "reward_std": 0.17731983959674835, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08909184485673904, "sampling/sampling_logp_difference/max": 1.0417542457580566, "sampling/importance_sampling_ratio/min": 0.35283517837524414, "sampling/importance_sampling_ratio/mean": 0.978323221206665, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.30892448872327805, "clip_ratio/low_mean": 0.007575757801532745, "clip_ratio/low_min": 0.007575757801532745, "clip_ratio/high_mean": 0.08523908769711852, "clip_ratio/high_max": 0.08523908769711852, "clip_ratio/region_mean": 0.09281484549865127, "reward_total_mean": 0.39515113830566406, "reward_meter_mean": 0.7480518817901611, "reward_meter_std": 0.29896217584609985, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9610674381256104, "reward_repeat_soft_std": 0.01786000281572342, "reward_judge_quality_mean": 0.17500001192092896, "reward_judge_quality_std": 0.11649646610021591, "reward_total_composite_mean": 0.39515113830566406, "reward_total_composite_std": 0.17731983959674835} {"timestamp_utc": "2026-04-13T12:36:08Z", "mode": "train", "global_step": 2136, "epoch": 0.21456554495228528, "loss": 0.0455, "grad_norm": 12.88821792602539, "learning_rate": 3.5303030303030304e-06, "num_tokens": 3842041.0, "completions/mean_length": 29.0, "completions/min_length": 18.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.0, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.8173210024833679, "rewards/meter/std": 0.33915388584136963, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9517419338226318, "rewards/repeat_soft/std": 0.020060529932379723, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.42245009541511536, "rewards/total_composite/std": 0.032370131462812424, "reward": 0.42245009541511536, "reward_std": 0.032370127737522125, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13327038288116455, "sampling/sampling_logp_difference/max": 2.3602185249328613, "sampling/importance_sampling_ratio/min": 0.094399593770504, "sampling/importance_sampling_ratio/mean": 0.9857088923454285, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.6100064590573311, "clip_ratio/low_mean": 0.018518518656492233, "clip_ratio/low_min": 0.018518518656492233, "clip_ratio/high_mean": 0.10606145393103361, "clip_ratio/high_max": 0.10606145393103361, "clip_ratio/region_mean": 0.12457997258752584, "reward_total_mean": 0.42245009541511536, "reward_meter_mean": 0.8173210024833679, "reward_meter_std": 0.33915388584136963, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9517419338226318, "reward_repeat_soft_std": 0.020060529932379723, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.42245009541511536, "reward_total_composite_std": 0.032370131462812424} {"timestamp_utc": "2026-04-13T12:36:15Z", "mode": "train", "global_step": 2137, "epoch": 0.21466599698643898, "loss": 0.0787, "grad_norm": 24.174345016479492, "learning_rate": 3.5272727272727276e-06, "num_tokens": 3843421.0, "completions/mean_length": 21.5, "completions/min_length": 15.0, "completions/max_length": 27.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.5, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.6113148927688599, "rewards/meter/std": 0.4718313217163086, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9569318294525146, "rewards/repeat_soft/std": 0.015749182552099228, "rewards/judge_quality/mean": 0.24625001847743988, "rewards/judge_quality/std": 0.2722361087799072, "rewards/total_composite/mean": 0.40339311957359314, "rewards/total_composite/std": 0.0453665629029274, "reward": 0.40339311957359314, "reward_std": 0.0453665629029274, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.14792943000793457, "sampling/sampling_logp_difference/max": 1.133669376373291, "sampling/importance_sampling_ratio/min": 0.32185009121894836, "sampling/importance_sampling_ratio/mean": 0.9756884574890137, "sampling/importance_sampling_ratio/max": 1.6885017156600952, "entropy": 0.930234007537365, "clip_ratio/low_mean": 0.03321759309619665, "clip_ratio/low_min": 0.03321759309619665, "clip_ratio/high_mean": 0.10881273122504354, "clip_ratio/high_max": 0.10881273122504354, "clip_ratio/region_mean": 0.1420303243212402, "reward_total_mean": 0.40339311957359314, "reward_meter_mean": 0.6113148927688599, "reward_meter_std": 0.4718313217163086, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9569318294525146, "reward_repeat_soft_std": 0.015749182552099228, "reward_judge_quality_mean": 0.24625001847743988, "reward_judge_quality_std": 0.2722361087799072, "reward_total_composite_mean": 0.40339311957359314, "reward_total_composite_std": 0.0453665629029274} {"timestamp_utc": "2026-04-13T12:36:22Z", "mode": "train", "global_step": 2138, "epoch": 0.21476644902059266, "loss": 0.009, "grad_norm": 6.568856716156006, "learning_rate": 3.5242424242424244e-06, "num_tokens": 3845705.0, "completions/mean_length": 118.5, "completions/min_length": 105.0, "completions/max_length": 134.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 118.5, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 134.0, "rewards/meter/mean": 0.8722131848335266, "rewards/meter/std": 0.14797870814800262, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8938058614730835, "rewards/repeat_soft/std": 0.03232758864760399, "rewards/judge_quality/mean": 0.17250001430511475, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.43214356899261475, "rewards/total_composite/std": 0.019955281168222427, "reward": 0.43214356899261475, "reward_std": 0.019955281168222427, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10581151396036148, "sampling/sampling_logp_difference/max": 2.4617185592651367, "sampling/importance_sampling_ratio/min": 0.08528824895620346, "sampling/importance_sampling_ratio/mean": 1.0000040531158447, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.43669046834111214, "clip_ratio/low_mean": 0.03886368125677109, "clip_ratio/low_min": 0.03886368125677109, "clip_ratio/high_mean": 0.057091874070465565, "clip_ratio/high_max": 0.057091874070465565, "clip_ratio/region_mean": 0.09595555532723665, "reward_total_mean": 0.43214356899261475, "reward_meter_mean": 0.8722131848335266, "reward_meter_std": 0.14797870814800262, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8938058614730835, "reward_repeat_soft_std": 0.03232758864760399, "reward_judge_quality_mean": 0.17250001430511475, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.43214356899261475, "reward_total_composite_std": 0.019955281168222427} {"timestamp_utc": "2026-04-13T12:36:28Z", "mode": "train", "global_step": 2139, "epoch": 0.21486690105474635, "loss": -0.0343, "grad_norm": 13.603361129760742, "learning_rate": 3.5212121212121213e-06, "num_tokens": 3847031.0, "completions/mean_length": 27.75, "completions/min_length": 24.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.75, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.9820600748062134, "rewards/meter/std": 0.023328015580773354, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9340013265609741, "rewards/repeat_soft/std": 0.030499953776597977, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4358510673046112, "rewards/total_composite/std": 0.0052992128767073154, "reward": 0.4358510673046112, "reward_std": 0.005299215670675039, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0838530957698822, "sampling/sampling_logp_difference/max": 1.0038328170776367, "sampling/importance_sampling_ratio/min": 0.36647212505340576, "sampling/importance_sampling_ratio/mean": 1.0212786197662354, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5129394680261612, "clip_ratio/low_mean": 0.05222701234742999, "clip_ratio/low_min": 0.05222701234742999, "clip_ratio/high_mean": 0.016964286100119352, "clip_ratio/high_max": 0.016964286100119352, "clip_ratio/region_mean": 0.06919129844754934, "reward_total_mean": 0.4358510673046112, "reward_meter_mean": 0.9820600748062134, "reward_meter_std": 0.023328015580773354, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9340013265609741, "reward_repeat_soft_std": 0.030499953776597977, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4358510673046112, "reward_total_composite_std": 0.0052992128767073154} {"timestamp_utc": "2026-04-13T12:36:35Z", "mode": "train", "global_step": 2140, "epoch": 0.21496735308890005, "loss": 0.0841, "grad_norm": 11.767311096191406, "learning_rate": 3.5181818181818185e-06, "num_tokens": 3848601.0, "completions/mean_length": 34.25, "completions/min_length": 27.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.25, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.8089805841445923, "rewards/meter/std": 0.216323122382164, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9660546183586121, "rewards/repeat_soft/std": 0.015586726367473602, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.4237838089466095, "rewards/total_composite/std": 0.02083503268659115, "reward": 0.4237838089466095, "reward_std": 0.020835043862462044, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08979345858097076, "sampling/sampling_logp_difference/max": 2.035578727722168, "sampling/importance_sampling_ratio/min": 0.13060487806797028, "sampling/importance_sampling_ratio/mean": 1.0170397758483887, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3702515698969364, "clip_ratio/low_mean": 0.028292838484048843, "clip_ratio/low_min": 0.028292838484048843, "clip_ratio/high_mean": 0.061627304181456566, "clip_ratio/high_max": 0.061627304181456566, "clip_ratio/region_mean": 0.08992014266550541, "reward_total_mean": 0.4237838089466095, "reward_meter_mean": 0.8089805841445923, "reward_meter_std": 0.216323122382164, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9660546183586121, "reward_repeat_soft_std": 0.015586726367473602, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.4237838089466095, "reward_total_composite_std": 0.02083503268659115} {"timestamp_utc": "2026-04-13T12:36:42Z", "mode": "train", "global_step": 2141, "epoch": 0.21506780512305373, "loss": 0.0321, "grad_norm": 7.464240550994873, "learning_rate": 3.5151515151515154e-06, "num_tokens": 3850272.0, "completions/mean_length": 52.875, "completions/min_length": 50.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.875, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9091874361038208, "rewards/meter/std": 0.1122494712471962, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9380810260772705, "rewards/repeat_soft/std": 0.0403120182454586, "rewards/judge_quality/mean": 0.1537500023841858, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.43157362937927246, "rewards/total_composite/std": 0.010991547256708145, "reward": 0.43157362937927246, "reward_std": 0.010991537012159824, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09349511563777924, "sampling/sampling_logp_difference/max": 1.4340959787368774, "sampling/importance_sampling_ratio/min": 0.23833072185516357, "sampling/importance_sampling_ratio/mean": 1.0083507299423218, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5642917603254318, "clip_ratio/low_mean": 0.09280007239431143, "clip_ratio/low_min": 0.09280007239431143, "clip_ratio/high_mean": 0.021226415410637856, "clip_ratio/high_max": 0.021226415410637856, "clip_ratio/region_mean": 0.11402648780494928, "reward_total_mean": 0.43157362937927246, "reward_meter_mean": 0.9091874361038208, "reward_meter_std": 0.1122494712471962, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9380810260772705, "reward_repeat_soft_std": 0.0403120182454586, "reward_judge_quality_mean": 0.1537500023841858, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.43157362937927246, "reward_total_composite_std": 0.010991547256708145} {"timestamp_utc": "2026-04-13T12:36:49Z", "mode": "train", "global_step": 2142, "epoch": 0.21516825715720744, "loss": -0.0837, "grad_norm": 8.290336608886719, "learning_rate": 3.512121212121212e-06, "num_tokens": 3852342.0, "completions/mean_length": 100.75, "completions/min_length": 77.0, "completions/max_length": 145.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 100.75, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.6486868858337402, "rewards/meter/std": 0.2888045907020569, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9150484800338745, "rewards/repeat_soft/std": 0.02585035189986229, "rewards/judge_quality/mean": 0.21000000834465027, "rewards/judge_quality/std": 0.08485280722379684, "rewards/total_composite/mean": 0.42709919810295105, "rewards/total_composite/std": 0.05602974444627762, "reward": 0.42709919810295105, "reward_std": 0.05602974444627762, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0966973677277565, "sampling/sampling_logp_difference/max": 1.9362258911132812, "sampling/importance_sampling_ratio/min": 0.14424732327461243, "sampling/importance_sampling_ratio/mean": 1.0121235847473145, "sampling/importance_sampling_ratio/max": 1.9052740335464478, "entropy": 0.5177117586135864, "clip_ratio/low_mean": 0.053842433262616396, "clip_ratio/low_min": 0.053842433262616396, "clip_ratio/high_mean": 0.058138023130595684, "clip_ratio/high_max": 0.058138023130595684, "clip_ratio/region_mean": 0.11198045639321208, "reward_total_mean": 0.42709919810295105, "reward_meter_mean": 0.6486868858337402, "reward_meter_std": 0.2888045907020569, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9150484800338745, "reward_repeat_soft_std": 0.02585035189986229, "reward_judge_quality_mean": 0.21000000834465027, "reward_judge_quality_std": 0.08485280722379684, "reward_total_composite_mean": 0.42709919810295105, "reward_total_composite_std": 0.05602974444627762} {"timestamp_utc": "2026-04-13T12:36:56Z", "mode": "train", "global_step": 2143, "epoch": 0.21526870919136112, "loss": 0.0574, "grad_norm": 7.681342124938965, "learning_rate": 3.509090909090909e-06, "num_tokens": 3854286.0, "completions/mean_length": 53.0, "completions/min_length": 49.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.0, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9523389339447021, "rewards/meter/std": 0.03752657771110535, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9398510456085205, "rewards/repeat_soft/std": 0.030083410441875458, "rewards/judge_quality/mean": 0.16500000655651093, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.4429652690887451, "rewards/total_composite/std": 0.008120769634842873, "reward": 0.4429652690887451, "reward_std": 0.008120759390294552, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09259400516748428, "sampling/sampling_logp_difference/max": 2.3451058864593506, "sampling/importance_sampling_ratio/min": 0.0958370566368103, "sampling/importance_sampling_ratio/mean": 1.0037641525268555, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.44840671494603157, "clip_ratio/low_mean": 0.04185404907912016, "clip_ratio/low_min": 0.04185404907912016, "clip_ratio/high_mean": 0.046734149334952235, "clip_ratio/high_max": 0.046734149334952235, "clip_ratio/region_mean": 0.0885881984140724, "reward_total_mean": 0.4429652690887451, "reward_meter_mean": 0.9523389339447021, "reward_meter_std": 0.03752657771110535, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9398510456085205, "reward_repeat_soft_std": 0.030083410441875458, "reward_judge_quality_mean": 0.16500000655651093, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.4429652690887451, "reward_total_composite_std": 0.008120769634842873} {"timestamp_utc": "2026-04-13T12:37:03Z", "mode": "train", "global_step": 2144, "epoch": 0.2153691612255148, "loss": -0.0546, "grad_norm": 8.980925559997559, "learning_rate": 3.5060606060606063e-06, "num_tokens": 3856427.0, "completions/mean_length": 78.625, "completions/min_length": 63.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.625, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.8787567019462585, "rewards/meter/std": 0.20320703089237213, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8748688697814941, "rewards/repeat_soft/std": 0.019511040300130844, "rewards/judge_quality/mean": 0.16124999523162842, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.4235439896583557, "rewards/total_composite/std": 0.02333262749016285, "reward": 0.4235439896583557, "reward_std": 0.0233326256275177, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.12014502286911011, "sampling/sampling_logp_difference/max": 1.8084826469421387, "sampling/importance_sampling_ratio/min": 0.1639026403427124, "sampling/importance_sampling_ratio/mean": 0.9930413365364075, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5412853918969631, "clip_ratio/low_mean": 0.023078530095517635, "clip_ratio/low_min": 0.023078530095517635, "clip_ratio/high_mean": 0.08712796587496996, "clip_ratio/high_max": 0.08712796587496996, "clip_ratio/region_mean": 0.1102064959704876, "reward_total_mean": 0.4235439896583557, "reward_meter_mean": 0.8787567019462585, "reward_meter_std": 0.20320703089237213, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8748688697814941, "reward_repeat_soft_std": 0.019511040300130844, "reward_judge_quality_mean": 0.16124999523162842, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.4235439896583557, "reward_total_composite_std": 0.02333262749016285} {"timestamp_utc": "2026-04-13T12:37:15Z", "mode": "train", "global_step": 2145, "epoch": 0.2154696132596685, "loss": -0.0907, "grad_norm": 3.2406935691833496, "learning_rate": 3.503030303030303e-06, "num_tokens": 3857950.0, "completions/mean_length": 86.375, "completions/min_length": 22.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 25.571430206298828, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.8375157713890076, "rewards/meter/std": 0.3385445475578308, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.8941828608512878, "rewards/repeat_soft/std": 0.11264170706272125, "rewards/judge_quality/mean": 0.13750000298023224, "rewards/judge_quality/std": 0.0353553406894207, "rewards/total_composite/mean": 0.36311087012290955, "rewards/total_composite/std": 0.15466636419296265, "reward": 0.36311087012290955, "reward_std": 0.15466636419296265, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11213835328817368, "sampling/sampling_logp_difference/max": 0.9431552886962891, "sampling/importance_sampling_ratio/min": 0.3893972635269165, "sampling/importance_sampling_ratio/mean": 1.052787184715271, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8141542971134186, "clip_ratio/low_mean": 0.021739130839705467, "clip_ratio/low_min": 0.021739130839705467, "clip_ratio/high_mean": 0.076064292807132, "clip_ratio/high_max": 0.076064292807132, "clip_ratio/region_mean": 0.09780342364683747, "reward_total_mean": 0.36311087012290955, "reward_meter_mean": 0.8375157713890076, "reward_meter_std": 0.3385445475578308, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.8941828608512878, "reward_repeat_soft_std": 0.11264170706272125, "reward_judge_quality_mean": 0.13750000298023224, "reward_judge_quality_std": 0.0353553406894207, "reward_total_composite_mean": 0.36311087012290955, "reward_total_composite_std": 0.15466636419296265} {"timestamp_utc": "2026-04-13T12:37:24Z", "mode": "train", "global_step": 2146, "epoch": 0.2155700652938222, "loss": 0.0478, "grad_norm": 6.35801362991333, "learning_rate": 3.5e-06, "num_tokens": 3860945.0, "completions/mean_length": 201.375, "completions/min_length": 176.0, "completions/max_length": 229.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 201.375, "completions/min_terminated_length": 176.0, "completions/max_terminated_length": 229.0, "rewards/meter/mean": 0.9370740652084351, "rewards/meter/std": 0.06427296251058578, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.05892555043101311, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8033504486083984, "rewards/repeat_soft/std": 0.07749520242214203, "rewards/judge_quality/mean": 0.13500000536441803, "rewards/judge_quality/std": 0.02777460776269436, "rewards/total_composite/mean": 0.365647554397583, "rewards/total_composite/std": 0.034212831407785416, "reward": 0.365647554397583, "reward_std": 0.03421282768249512, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08685857802629471, "sampling/sampling_logp_difference/max": 3.345730781555176, "sampling/importance_sampling_ratio/min": 0.03523445874452591, "sampling/importance_sampling_ratio/mean": 1.001274585723877, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.4255397655069828, "clip_ratio/low_mean": 0.03247932391241193, "clip_ratio/low_min": 0.03247932391241193, "clip_ratio/high_mean": 0.048446863889694214, "clip_ratio/high_max": 0.048446863889694214, "clip_ratio/region_mean": 0.08092618780210614, "reward_total_mean": 0.365647554397583, "reward_meter_mean": 0.9370740652084351, "reward_meter_std": 0.06427296251058578, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.05892555043101311, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8033504486083984, "reward_repeat_soft_std": 0.07749520242214203, "reward_judge_quality_mean": 0.13500000536441803, "reward_judge_quality_std": 0.02777460776269436, "reward_total_composite_mean": 0.365647554397583, "reward_total_composite_std": 0.034212831407785416} {"timestamp_utc": "2026-04-13T12:37:30Z", "mode": "train", "global_step": 2147, "epoch": 0.2156705173279759, "loss": -0.009, "grad_norm": 13.747391700744629, "learning_rate": 3.496969696969697e-06, "num_tokens": 3862703.0, "completions/mean_length": 45.75, "completions/min_length": 37.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.75, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.8913014531135559, "rewards/meter/std": 0.22273597121238708, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9591418504714966, "rewards/repeat_soft/std": 0.035527247935533524, "rewards/judge_quality/mean": 0.14625000953674316, "rewards/judge_quality/std": 0.010606604628264904, "rewards/total_composite/mean": 0.4283779263496399, "rewards/total_composite/std": 0.024290764704346657, "reward": 0.4283779263496399, "reward_std": 0.02429075725376606, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09208957105875015, "sampling/sampling_logp_difference/max": 1.9488413333892822, "sampling/importance_sampling_ratio/min": 0.14243902266025543, "sampling/importance_sampling_ratio/mean": 1.0007129907608032, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.44552335888147354, "clip_ratio/low_mean": 0.014354674611240625, "clip_ratio/low_min": 0.014354674611240625, "clip_ratio/high_mean": 0.08540754113346338, "clip_ratio/high_max": 0.08540754113346338, "clip_ratio/region_mean": 0.09976221574470401, "reward_total_mean": 0.4283779263496399, "reward_meter_mean": 0.8913014531135559, "reward_meter_std": 0.22273597121238708, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9591418504714966, "reward_repeat_soft_std": 0.035527247935533524, "reward_judge_quality_mean": 0.14625000953674316, "reward_judge_quality_std": 0.010606604628264904, "reward_total_composite_mean": 0.4283779263496399, "reward_total_composite_std": 0.024290764704346657} {"timestamp_utc": "2026-04-13T12:37:37Z", "mode": "train", "global_step": 2148, "epoch": 0.21577096936212958, "loss": -0.0048, "grad_norm": 12.051156997680664, "learning_rate": 3.493939393939394e-06, "num_tokens": 3864702.0, "completions/mean_length": 69.875, "completions/min_length": 62.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.875, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.8386724591255188, "rewards/meter/std": 0.2629724144935608, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9236710071563721, "rewards/repeat_soft/std": 0.02468961849808693, "rewards/judge_quality/mean": 0.1537500023841858, "rewards/judge_quality/std": 0.01922610215842724, "rewards/total_composite/mean": 0.37210574746131897, "rewards/total_composite/std": 0.15347681939601898, "reward": 0.37210574746131897, "reward_std": 0.1534768044948578, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10458245873451233, "sampling/sampling_logp_difference/max": 2.363548517227173, "sampling/importance_sampling_ratio/min": 0.09408576041460037, "sampling/importance_sampling_ratio/mean": 1.0113861560821533, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5813146941363811, "clip_ratio/low_mean": 0.018656716216355562, "clip_ratio/low_min": 0.018656716216355562, "clip_ratio/high_mean": 0.06361279031261802, "clip_ratio/high_max": 0.06361279031261802, "clip_ratio/region_mean": 0.08226950652897358, "reward_total_mean": 0.37210574746131897, "reward_meter_mean": 0.8386724591255188, "reward_meter_std": 0.2629724144935608, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9236710071563721, "reward_repeat_soft_std": 0.02468961849808693, "reward_judge_quality_mean": 0.1537500023841858, "reward_judge_quality_std": 0.01922610215842724, "reward_total_composite_mean": 0.37210574746131897, "reward_total_composite_std": 0.15347681939601898} {"timestamp_utc": "2026-04-13T12:37:43Z", "mode": "train", "global_step": 2149, "epoch": 0.21587142139628326, "loss": 0.0132, "grad_norm": 11.868727684020996, "learning_rate": 3.4909090909090913e-06, "num_tokens": 3866315.0, "completions/mean_length": 42.625, "completions/min_length": 41.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.625, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.9211106896400452, "rewards/meter/std": 0.10339754819869995, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9673133492469788, "rewards/repeat_soft/std": 0.019251564517617226, "rewards/judge_quality/mean": 0.1537500023841858, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.4373164772987366, "rewards/total_composite/std": 0.012059218250215054, "reward": 0.4373164772987366, "reward_std": 0.012059218250215054, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11840193718671799, "sampling/sampling_logp_difference/max": 1.9005436897277832, "sampling/importance_sampling_ratio/min": 0.14948733150959015, "sampling/importance_sampling_ratio/mean": 0.987741231918335, "sampling/importance_sampling_ratio/max": 1.8737012147903442, "entropy": 0.5505771040916443, "clip_ratio/low_mean": 0.0335365841165185, "clip_ratio/low_min": 0.0335365841165185, "clip_ratio/high_mean": 0.07705757208168507, "clip_ratio/high_max": 0.07705757208168507, "clip_ratio/region_mean": 0.11059415619820356, "reward_total_mean": 0.4373164772987366, "reward_meter_mean": 0.9211106896400452, "reward_meter_std": 0.10339754819869995, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9673133492469788, "reward_repeat_soft_std": 0.019251564517617226, "reward_judge_quality_mean": 0.1537500023841858, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.4373164772987366, "reward_total_composite_std": 0.012059218250215054} {"timestamp_utc": "2026-04-13T12:37:50Z", "mode": "train", "global_step": 2150, "epoch": 0.21597187343043697, "loss": -0.0127, "grad_norm": 11.236807823181152, "learning_rate": 3.4878787878787885e-06, "num_tokens": 3868253.0, "completions/mean_length": 52.25, "completions/min_length": 45.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.25, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.8498669862747192, "rewards/meter/std": 0.14511364698410034, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9805920124053955, "rewards/repeat_soft/std": 0.020208818838000298, "rewards/judge_quality/mean": 0.15000000596046448, "rewards/judge_quality/std": 0.01603567786514759, "rewards/total_composite/mean": 0.4297049641609192, "rewards/total_composite/std": 0.013732438907027245, "reward": 0.4297049641609192, "reward_std": 0.013732440769672394, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11804606765508652, "sampling/sampling_logp_difference/max": 2.0987229347229004, "sampling/importance_sampling_ratio/min": 0.12261290848255157, "sampling/importance_sampling_ratio/mean": 0.9951035976409912, "sampling/importance_sampling_ratio/max": 1.9407857656478882, "entropy": 0.589105699211359, "clip_ratio/low_mean": 0.03481538034975529, "clip_ratio/low_min": 0.03481538034975529, "clip_ratio/high_mean": 0.05818670894950628, "clip_ratio/high_max": 0.05818670894950628, "clip_ratio/region_mean": 0.09300208929926157, "reward_total_mean": 0.4297049641609192, "reward_meter_mean": 0.8498669862747192, "reward_meter_std": 0.14511364698410034, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9805920124053955, "reward_repeat_soft_std": 0.020208818838000298, "reward_judge_quality_mean": 0.15000000596046448, "reward_judge_quality_std": 0.01603567786514759, "reward_total_composite_mean": 0.4297049641609192, "reward_total_composite_std": 0.013732438907027245} {"timestamp_utc": "2026-04-13T12:38:37Z", "mode": "eval", "global_step": 2150, "epoch": 0.21597187343043697, "eval_loss": NaN, "eval_runtime": 46.3808, "eval_samples_per_second": 1.725, "eval_steps_per_second": 0.216, "eval_num_tokens": 3868253.0, "eval_completions/mean_length": 84.7, "eval_completions/min_length": 31.7, "eval_completions/max_length": 175.2, "eval_completions/clipped_ratio": 0.0125, "eval_completions/mean_terminated_length": 79.49821472167969, "eval_completions/min_terminated_length": 31.7, "eval_completions/max_terminated_length": 139.9, "eval_rewards/meter/mean": 0.7940246820449829, "eval_rewards/meter/std": 0.26315709128975867, "eval_rewards/count_adherence/mean": 0.9881250023841858, "eval_rewards/count_adherence/std": 0.03358757160604, "eval_rewards/hard_gate/mean": 0.975, "eval_rewards/hard_gate/std": 0.07071067690849304, "eval_rewards/repeat_soft/mean": 0.8970467984676361, "eval_rewards/repeat_soft/std": 0.07692538686096669, "eval_rewards/judge_quality/mean": 0.15700000524520874, "eval_rewards/judge_quality/std": 0.02237090365961194, "eval_rewards/total_composite/mean": 0.4048567622900009, "eval_rewards/total_composite/std": 0.0588286342099309, "eval_reward": 0.4048567622900009, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.046076252683997156, "eval_sampling/sampling_logp_difference/max": 0.909369945526123, "eval_sampling/importance_sampling_ratio/min": 0.418349090218544, "eval_sampling/importance_sampling_ratio/mean": 1.0083821773529054, "eval_sampling/importance_sampling_ratio/max": 1.4984214782714844, "eval_entropy": 0.46796957552433016, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.4048567622900009, "eval_reward_meter_mean": 0.7940246820449829, "eval_reward_meter_std": 0.26315709128975867, "eval_reward_count_adherence_mean": 0.9881250023841858, "eval_reward_count_adherence_std": 0.03358757160604, "eval_reward_hard_gate_mean": 0.975, "eval_reward_hard_gate_std": 0.07071067690849304, "eval_reward_repeat_soft_mean": 0.8970467984676361, "eval_reward_repeat_soft_std": 0.07692538686096669, "eval_reward_judge_quality_mean": 0.15700000524520874, "eval_reward_judge_quality_std": 0.02237090365961194, "eval_reward_total_composite_mean": 0.4048567622900009, "eval_reward_total_composite_std": 0.0588286342099309} {"timestamp_utc": "2026-04-13T12:38:47Z", "mode": "train", "global_step": 2151, "epoch": 0.21607232546459065, "loss": 0.0412, "grad_norm": 7.0074052810668945, "learning_rate": 3.4848484848484854e-06, "num_tokens": 3870707.0, "completions/mean_length": 108.75, "completions/min_length": 96.0, "completions/max_length": 118.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.75, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 118.0, "rewards/meter/mean": 0.9364088773727417, "rewards/meter/std": 0.06607461720705032, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.8666599988937378, "rewards/repeat_soft/std": 0.04671601206064224, "rewards/judge_quality/mean": 0.1612500101327896, "rewards/judge_quality/std": 0.02748376689851284, "rewards/total_composite/mean": 0.42774778604507446, "rewards/total_composite/std": 0.020854637026786804, "reward": 0.42774778604507446, "reward_std": 0.020854640752077103, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.092539943754673, "sampling/sampling_logp_difference/max": 3.6278061866760254, "sampling/importance_sampling_ratio/min": 0.02657441981136799, "sampling/importance_sampling_ratio/mean": 1.0023683309555054, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3217124678194523, "clip_ratio/low_mean": 0.030726388096809387, "clip_ratio/low_min": 0.030726388096809387, "clip_ratio/high_mean": 0.05013436125591397, "clip_ratio/high_max": 0.05013436125591397, "clip_ratio/region_mean": 0.08086074935272336, "reward_total_mean": 0.42774778604507446, "reward_meter_mean": 0.9364088773727417, "reward_meter_std": 0.06607461720705032, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.8666599988937378, "reward_repeat_soft_std": 0.04671601206064224, "reward_judge_quality_mean": 0.1612500101327896, "reward_judge_quality_std": 0.02748376689851284, "reward_total_composite_mean": 0.42774778604507446, "reward_total_composite_std": 0.020854637026786804}