{"timestamp_utc": "2026-04-13T12:45:48Z", "mode": "train", "global_step": 1, "epoch": 0.00010045203415369161, "loss": 0.1244, "grad_norm": 20.99456787109375, "learning_rate": 1e-05, "num_tokens": 1670.0, "completions/mean_length": 41.75, "completions/min_length": 29.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.75, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.6725290417671204, "rewards/meter/std": 0.40340036153793335, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9957022070884705, "rewards/repeat_soft/std": 0.00409209867939353, "rewards/judge_quality/mean": 0.26375001668930054, "rewards/judge_quality/std": 0.0645727664232254, "rewards/total_composite/mean": 0.2045568972826004, "rewards/total_composite/std": 0.10483554750680923, "reward": 0.2045568972826004, "reward_std": 0.10483554750680923, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22603146731853485, "sampling/sampling_logp_difference/max": 1.5626206398010254, "sampling/importance_sampling_ratio/min": 0.20958609879016876, "sampling/importance_sampling_ratio/mean": 1.013262391090393, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.01923768222332, "clip_ratio/low_mean": 0.10917956382036209, "clip_ratio/low_min": 0.10917956382036209, "clip_ratio/high_mean": 0.12637650407850742, "clip_ratio/high_max": 0.12637650407850742, "clip_ratio/region_mean": 0.23555606789886951, "reward_total_mean": 0.2045568972826004, "reward_meter_mean": 0.6725290417671204, "reward_meter_std": 0.40340036153793335, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9957022070884705, "reward_repeat_soft_std": 0.00409209867939353, "reward_judge_quality_mean": 0.26375001668930054, "reward_judge_quality_std": 0.0645727664232254, "reward_total_composite_mean": 0.2045568972826004, "reward_total_composite_std": 0.10483554750680923} {"timestamp_utc": "2026-04-13T12:45:56Z", "mode": "train", "global_step": 2, "epoch": 0.00020090406830738323, "loss": 0.0213, "grad_norm": 7.792138576507568, "learning_rate": 9.996969696969698e-06, "num_tokens": 4134.0, "completions/mean_length": 144.0, "completions/min_length": 103.0, "completions/max_length": 163.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 144.0, "completions/min_terminated_length": 103.0, "completions/max_terminated_length": 163.0, "rewards/meter/mean": 0.6690956950187683, "rewards/meter/std": 0.369981974363327, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9968761205673218, "rewards/repeat_soft/std": 0.0025736435782164335, "rewards/judge_quality/mean": 0.3100000023841858, "rewards/judge_quality/std": 0.14172407984733582, "rewards/total_composite/mean": 0.24503539502620697, "rewards/total_composite/std": 0.15705519914627075, "reward": 0.24503539502620697, "reward_std": 0.15705521404743195, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19617469608783722, "sampling/sampling_logp_difference/max": 2.204803943634033, "sampling/importance_sampling_ratio/min": 0.11027213931083679, "sampling/importance_sampling_ratio/mean": 1.0382649898529053, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9801125675439835, "clip_ratio/low_mean": 0.16077206656336784, "clip_ratio/low_min": 0.16077206656336784, "clip_ratio/high_mean": 0.04601428098976612, "clip_ratio/high_max": 0.04601428098976612, "clip_ratio/region_mean": 0.20678634755313396, "reward_total_mean": 0.24503539502620697, "reward_meter_mean": 0.6690956950187683, "reward_meter_std": 0.369981974363327, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9968761205673218, "reward_repeat_soft_std": 0.0025736435782164335, "reward_judge_quality_mean": 0.3100000023841858, "reward_judge_quality_std": 0.14172407984733582, "reward_total_composite_mean": 0.24503539502620697, "reward_total_composite_std": 0.15705519914627075} {"timestamp_utc": "2026-04-13T12:46:02Z", "mode": "train", "global_step": 3, "epoch": 0.00030135610246107485, "loss": -0.1053, "grad_norm": 23.21904754638672, "learning_rate": 9.993939393939395e-06, "num_tokens": 5844.0, "completions/mean_length": 36.75, "completions/min_length": 27.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.75, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.6859860420227051, "rewards/meter/std": 0.3346250057220459, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9924981594085693, "rewards/repeat_soft/std": 0.010377585887908936, "rewards/judge_quality/mean": 0.3062500059604645, "rewards/judge_quality/std": 0.14302222430706024, "rewards/total_composite/mean": 0.23595881462097168, "rewards/total_composite/std": 0.10526535660028458, "reward": 0.23595881462097168, "reward_std": 0.10526534914970398, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.3009033203125, "sampling/sampling_logp_difference/max": 4.2564239501953125, "sampling/importance_sampling_ratio/min": 0.014172895811498165, "sampling/importance_sampling_ratio/mean": 0.9777875542640686, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2389525398612022, "clip_ratio/low_mean": 0.1306113675236702, "clip_ratio/low_min": 0.1306113675236702, "clip_ratio/high_mean": 0.10717256367206573, "clip_ratio/high_max": 0.10717256367206573, "clip_ratio/region_mean": 0.23778393119573593, "reward_total_mean": 0.23595881462097168, "reward_meter_mean": 0.6859860420227051, "reward_meter_std": 0.3346250057220459, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9924981594085693, "reward_repeat_soft_std": 0.010377585887908936, "reward_judge_quality_mean": 0.3062500059604645, "reward_judge_quality_std": 0.14302222430706024, "reward_total_composite_mean": 0.23595881462097168, "reward_total_composite_std": 0.10526535660028458} {"timestamp_utc": "2026-04-13T12:46:08Z", "mode": "train", "global_step": 4, "epoch": 0.00040180813661476645, "loss": -0.0566, "grad_norm": 16.3281192779541, "learning_rate": 9.990909090909093e-06, "num_tokens": 7454.0, "completions/mean_length": 44.25, "completions/min_length": 30.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.25, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.4112703204154968, "rewards/meter/std": 0.37593206763267517, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.997799277305603, "rewards/repeat_soft/std": 0.0031309700571000576, "rewards/judge_quality/mean": 0.3812499940395355, "rewards/judge_quality/std": 0.166770800948143, "rewards/total_composite/mean": 0.23073405027389526, "rewards/total_composite/std": 0.11460831761360168, "reward": 0.23073405027389526, "reward_std": 0.11460831761360168, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20872721076011658, "sampling/sampling_logp_difference/max": 1.4540247917175293, "sampling/importance_sampling_ratio/min": 0.23362809419631958, "sampling/importance_sampling_ratio/mean": 1.02910578250885, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8074360936880112, "clip_ratio/low_mean": 0.08966553024947643, "clip_ratio/low_min": 0.08966553024947643, "clip_ratio/high_mean": 0.1273432057350874, "clip_ratio/high_max": 0.1273432057350874, "clip_ratio/region_mean": 0.21700873598456383, "reward_total_mean": 0.23073405027389526, "reward_meter_mean": 0.4112703204154968, "reward_meter_std": 0.37593206763267517, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.997799277305603, "reward_repeat_soft_std": 0.0031309700571000576, "reward_judge_quality_mean": 0.3812499940395355, "reward_judge_quality_std": 0.166770800948143, "reward_total_composite_mean": 0.23073405027389526, "reward_total_composite_std": 0.11460831761360168} {"timestamp_utc": "2026-04-13T12:46:15Z", "mode": "train", "global_step": 5, "epoch": 0.0005022601707684581, "loss": 0.0491, "grad_norm": 8.495488166809082, "learning_rate": 9.987878787878788e-06, "num_tokens": 9877.0, "completions/mean_length": 125.875, "completions/min_length": 101.0, "completions/max_length": 153.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 125.875, "completions/min_terminated_length": 101.0, "completions/max_terminated_length": 153.0, "rewards/meter/mean": 0.5144270658493042, "rewards/meter/std": 0.4267750382423401, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9954259395599365, "rewards/repeat_soft/std": 0.0038400308694690466, "rewards/judge_quality/mean": 0.30250000953674316, "rewards/judge_quality/std": 0.20954032242298126, "rewards/total_composite/mean": 0.19073012471199036, "rewards/total_composite/std": 0.1002202183008194, "reward": 0.19073012471199036, "reward_std": 0.1002202108502388, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18132857978343964, "sampling/sampling_logp_difference/max": 2.1783599853515625, "sampling/importance_sampling_ratio/min": 0.11322707682847977, "sampling/importance_sampling_ratio/mean": 1.0050190687179565, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3385392278432846, "clip_ratio/low_mean": 0.07284662686288357, "clip_ratio/low_min": 0.07284662686288357, "clip_ratio/high_mean": 0.11934021860361099, "clip_ratio/high_max": 0.11934021860361099, "clip_ratio/region_mean": 0.19218684546649456, "reward_total_mean": 0.19073012471199036, "reward_meter_mean": 0.5144270658493042, "reward_meter_std": 0.4267750382423401, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9954259395599365, "reward_repeat_soft_std": 0.0038400308694690466, "reward_judge_quality_mean": 0.30250000953674316, "reward_judge_quality_std": 0.20954032242298126, "reward_total_composite_mean": 0.19073012471199036, "reward_total_composite_std": 0.1002202183008194} {"timestamp_utc": "2026-04-13T12:46:23Z", "mode": "train", "global_step": 6, "epoch": 0.0006027122049221497, "loss": 0.0912, "grad_norm": 9.133028030395508, "learning_rate": 9.984848484848485e-06, "num_tokens": 12497.0, "completions/mean_length": 128.5, "completions/min_length": 84.0, "completions/max_length": 175.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 128.5, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 175.0, "rewards/meter/mean": 0.9546395540237427, "rewards/meter/std": 0.10463159531354904, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.998668909072876, "rewards/repeat_soft/std": 0.001581631600856781, "rewards/judge_quality/mean": 0.23875001072883606, "rewards/judge_quality/std": 0.027483763173222542, "rewards/total_composite/mean": 0.22964996099472046, "rewards/total_composite/std": 0.026852471753954887, "reward": 0.22964996099472046, "reward_std": 0.026852469891309738, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2118406891822815, "sampling/sampling_logp_difference/max": 1.8763961791992188, "sampling/importance_sampling_ratio/min": 0.15314100682735443, "sampling/importance_sampling_ratio/mean": 1.0181084871292114, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.359899118542671, "clip_ratio/low_mean": 0.13176005519926548, "clip_ratio/low_min": 0.13176005519926548, "clip_ratio/high_mean": 0.054821427911520004, "clip_ratio/high_max": 0.054821427911520004, "clip_ratio/region_mean": 0.18658148311078548, "reward_total_mean": 0.22964996099472046, "reward_meter_mean": 0.9546395540237427, "reward_meter_std": 0.10463159531354904, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.998668909072876, "reward_repeat_soft_std": 0.001581631600856781, "reward_judge_quality_mean": 0.23875001072883606, "reward_judge_quality_std": 0.027483763173222542, "reward_total_composite_mean": 0.22964996099472046, "reward_total_composite_std": 0.026852471753954887} {"timestamp_utc": "2026-04-13T12:46:30Z", "mode": "train", "global_step": 7, "epoch": 0.0007031642390758413, "loss": 0.0142, "grad_norm": 11.58638858795166, "learning_rate": 9.981818181818183e-06, "num_tokens": 15039.0, "completions/mean_length": 133.75, "completions/min_length": 85.0, "completions/max_length": 167.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 133.75, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 167.0, "rewards/meter/mean": 0.45653659105300903, "rewards/meter/std": 0.416522741317749, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9975261092185974, "rewards/repeat_soft/std": 0.002966292668133974, "rewards/judge_quality/mean": 0.23499999940395355, "rewards/judge_quality/std": 0.02777460403740406, "rewards/total_composite/mean": 0.14990538358688354, "rewards/total_composite/std": 0.05683733522891998, "reward": 0.14990538358688354, "reward_std": 0.056837331503629684, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21259059011936188, "sampling/sampling_logp_difference/max": 1.5976266860961914, "sampling/importance_sampling_ratio/min": 0.20237626135349274, "sampling/importance_sampling_ratio/mean": 1.0176150798797607, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.273042529821396, "clip_ratio/low_mean": 0.12764834240078926, "clip_ratio/low_min": 0.12764834240078926, "clip_ratio/high_mean": 0.09455128386616707, "clip_ratio/high_max": 0.09455128386616707, "clip_ratio/region_mean": 0.22219962626695633, "reward_total_mean": 0.14990538358688354, "reward_meter_mean": 0.45653659105300903, "reward_meter_std": 0.416522741317749, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9975261092185974, "reward_repeat_soft_std": 0.002966292668133974, "reward_judge_quality_mean": 0.23499999940395355, "reward_judge_quality_std": 0.02777460403740406, "reward_total_composite_mean": 0.14990538358688354, "reward_total_composite_std": 0.05683733522891998} {"timestamp_utc": "2026-04-13T12:46:36Z", "mode": "train", "global_step": 8, "epoch": 0.0008036162732295329, "loss": -0.0058, "grad_norm": 30.844223022460938, "learning_rate": 9.97878787878788e-06, "num_tokens": 16480.0, "completions/mean_length": 24.125, "completions/min_length": 18.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.125, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.905945897102356, "rewards/meter/std": 0.1110290139913559, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9606921672821045, "rewards/repeat_soft/std": 0.005113361403346062, "rewards/judge_quality/mean": 0.7437499761581421, "rewards/judge_quality/std": 0.2624574899673462, "rewards/total_composite/mean": 0.6936631202697754, "rewards/total_composite/std": 0.2499092072248459, "reward": 0.6936631202697754, "reward_std": 0.2499092072248459, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18403935432434082, "sampling/sampling_logp_difference/max": 2.2623558044433594, "sampling/importance_sampling_ratio/min": 0.10410494357347488, "sampling/importance_sampling_ratio/mean": 1.0264980792999268, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5599659457802773, "clip_ratio/low_mean": 0.06472222320735455, "clip_ratio/low_min": 0.06472222320735455, "clip_ratio/high_mean": 0.10283747036010027, "clip_ratio/high_max": 0.10283747036010027, "clip_ratio/region_mean": 0.16755969356745481, "reward_total_mean": 0.6936631202697754, "reward_meter_mean": 0.905945897102356, "reward_meter_std": 0.1110290139913559, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9606921672821045, "reward_repeat_soft_std": 0.005113361403346062, "reward_judge_quality_mean": 0.7437499761581421, "reward_judge_quality_std": 0.2624574899673462, "reward_total_composite_mean": 0.6936631202697754, "reward_total_composite_std": 0.2499092072248459} {"timestamp_utc": "2026-04-13T12:46:45Z", "mode": "train", "global_step": 9, "epoch": 0.0009040683073832245, "loss": 0.1804, "grad_norm": 7.546041488647461, "learning_rate": 9.975757575757577e-06, "num_tokens": 19319.0, "completions/mean_length": 148.875, "completions/min_length": 98.0, "completions/max_length": 220.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 148.875, "completions/min_terminated_length": 98.0, "completions/max_terminated_length": 220.0, "rewards/meter/mean": 0.9833512306213379, "rewards/meter/std": 0.016487499698996544, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9962527751922607, "rewards/repeat_soft/std": 0.0033018917310982943, "rewards/judge_quality/mean": 0.26750001311302185, "rewards/judge_quality/std": 0.06840008497238159, "rewards/total_composite/mean": 0.2634543478488922, "rewards/total_composite/std": 0.06922971457242966, "reward": 0.2634543478488922, "reward_std": 0.06922971457242966, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2024044394493103, "sampling/sampling_logp_difference/max": 1.5941190719604492, "sampling/importance_sampling_ratio/min": 0.20308735966682434, "sampling/importance_sampling_ratio/mean": 1.0357457399368286, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.254930019378662, "clip_ratio/low_mean": 0.08165346831083298, "clip_ratio/low_min": 0.08165346831083298, "clip_ratio/high_mean": 0.09651176072657108, "clip_ratio/high_max": 0.09651176072657108, "clip_ratio/region_mean": 0.17816522903740406, "reward_total_mean": 0.2634543478488922, "reward_meter_mean": 0.9833512306213379, "reward_meter_std": 0.016487499698996544, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9962527751922607, "reward_repeat_soft_std": 0.0033018917310982943, "reward_judge_quality_mean": 0.26750001311302185, "reward_judge_quality_std": 0.06840008497238159, "reward_total_composite_mean": 0.2634543478488922, "reward_total_composite_std": 0.06922971457242966} {"timestamp_utc": "2026-04-13T12:46:52Z", "mode": "train", "global_step": 10, "epoch": 0.0010045203415369162, "loss": 0.1814, "grad_norm": 20.30479621887207, "learning_rate": 9.972727272727274e-06, "num_tokens": 20869.0, "completions/mean_length": 31.75, "completions/min_length": 21.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.75, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.7482462525367737, "rewards/meter/std": 0.40584293007850647, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9827924370765686, "rewards/repeat_soft/std": 0.029328446835279465, "rewards/judge_quality/mean": 0.2524999976158142, "rewards/judge_quality/std": 0.06902380287647247, "rewards/total_composite/mean": 0.21522781252861023, "rewards/total_composite/std": 0.10106831789016724, "reward": 0.21522781252861023, "reward_std": 0.10106831789016724, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22601714730262756, "sampling/sampling_logp_difference/max": 1.5818405151367188, "sampling/importance_sampling_ratio/min": 0.20559634268283844, "sampling/importance_sampling_ratio/mean": 1.03934907913208, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5930405110120773, "clip_ratio/low_mean": 0.08459400478750467, "clip_ratio/low_min": 0.08459400478750467, "clip_ratio/high_mean": 0.15337539464235306, "clip_ratio/high_max": 0.15337539464235306, "clip_ratio/region_mean": 0.23796939942985773, "reward_total_mean": 0.21522781252861023, "reward_meter_mean": 0.7482462525367737, "reward_meter_std": 0.40584293007850647, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9827924370765686, "reward_repeat_soft_std": 0.029328446835279465, "reward_judge_quality_mean": 0.2524999976158142, "reward_judge_quality_std": 0.06902380287647247, "reward_total_composite_mean": 0.21522781252861023, "reward_total_composite_std": 0.10106831789016724} {"timestamp_utc": "2026-04-13T12:46:58Z", "mode": "train", "global_step": 11, "epoch": 0.0011049723756906078, "loss": 0.177, "grad_norm": 13.777490615844727, "learning_rate": 9.96969696969697e-06, "num_tokens": 22443.0, "completions/mean_length": 46.75, "completions/min_length": 33.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.75, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.5961973667144775, "rewards/meter/std": 0.4163079261779785, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9969432353973389, "rewards/repeat_soft/std": 0.004917121026664972, "rewards/judge_quality/mean": 0.32749998569488525, "rewards/judge_quality/std": 0.14606750011444092, "rewards/total_composite/mean": 0.2538233697414398, "rewards/total_composite/std": 0.17784301936626434, "reward": 0.2538233697414398, "reward_std": 0.17784300446510315, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2327873557806015, "sampling/sampling_logp_difference/max": 2.384127616882324, "sampling/importance_sampling_ratio/min": 0.09216935187578201, "sampling/importance_sampling_ratio/mean": 1.0264379978179932, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2318453043699265, "clip_ratio/low_mean": 0.14783670287579298, "clip_ratio/low_min": 0.14783670287579298, "clip_ratio/high_mean": 0.05930736102163792, "clip_ratio/high_max": 0.05930736102163792, "clip_ratio/region_mean": 0.2071440638974309, "reward_total_mean": 0.2538233697414398, "reward_meter_mean": 0.5961973667144775, "reward_meter_std": 0.4163079261779785, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9969432353973389, "reward_repeat_soft_std": 0.004917121026664972, "reward_judge_quality_mean": 0.32749998569488525, "reward_judge_quality_std": 0.14606750011444092, "reward_total_composite_mean": 0.2538233697414398, "reward_total_composite_std": 0.17784301936626434} {"timestamp_utc": "2026-04-13T12:47:04Z", "mode": "train", "global_step": 12, "epoch": 0.0012054244098442994, "loss": 0.0963, "grad_norm": 21.680850982666016, "learning_rate": 9.966666666666667e-06, "num_tokens": 23907.0, "completions/mean_length": 23.0, "completions/min_length": 15.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.0, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.46723121404647827, "rewards/meter/std": 0.4440421462059021, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9610613584518433, "rewards/repeat_soft/std": 0.004068940877914429, "rewards/judge_quality/mean": 0.6775000095367432, "rewards/judge_quality/std": 0.2979333698749542, "rewards/total_composite/mean": 0.4028964340686798, "rewards/total_composite/std": 0.25024446845054626, "reward": 0.4028964340686798, "reward_std": 0.25024449825286865, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23249541223049164, "sampling/sampling_logp_difference/max": 1.4548969268798828, "sampling/importance_sampling_ratio/min": 0.23342442512512207, "sampling/importance_sampling_ratio/mean": 1.039197325706482, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.634026363492012, "clip_ratio/low_mean": 0.12728216219693422, "clip_ratio/low_min": 0.12728216219693422, "clip_ratio/high_mean": 0.11459790356457233, "clip_ratio/high_max": 0.11459790356457233, "clip_ratio/region_mean": 0.24188006576150656, "reward_total_mean": 0.4028964340686798, "reward_meter_mean": 0.46723121404647827, "reward_meter_std": 0.4440421462059021, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9610613584518433, "reward_repeat_soft_std": 0.004068940877914429, "reward_judge_quality_mean": 0.6775000095367432, "reward_judge_quality_std": 0.2979333698749542, "reward_total_composite_mean": 0.4028964340686798, "reward_total_composite_std": 0.25024446845054626} {"timestamp_utc": "2026-04-13T12:47:10Z", "mode": "train", "global_step": 13, "epoch": 0.001305876443997991, "loss": 0.0782, "grad_norm": 20.452024459838867, "learning_rate": 9.963636363636364e-06, "num_tokens": 25545.0, "completions/mean_length": 47.75, "completions/min_length": 30.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.75, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.5705592632293701, "rewards/meter/std": 0.37744197249412537, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9955551624298096, "rewards/repeat_soft/std": 0.0066154426895082, "rewards/judge_quality/mean": 0.2524999976158142, "rewards/judge_quality/std": 0.06902380287647247, "rewards/total_composite/mean": 0.17487716674804688, "rewards/total_composite/std": 0.051213208585977554, "reward": 0.17487716674804688, "reward_std": 0.05121320113539696, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22743971645832062, "sampling/sampling_logp_difference/max": 2.7528858184814453, "sampling/importance_sampling_ratio/min": 0.06374364346265793, "sampling/importance_sampling_ratio/mean": 1.0200012922286987, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6435254663228989, "clip_ratio/low_mean": 0.08463804796338081, "clip_ratio/low_min": 0.08463804796338081, "clip_ratio/high_mean": 0.14737549424171448, "clip_ratio/high_max": 0.14737549424171448, "clip_ratio/region_mean": 0.2320135422050953, "reward_total_mean": 0.17487716674804688, "reward_meter_mean": 0.5705592632293701, "reward_meter_std": 0.37744197249412537, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9955551624298096, "reward_repeat_soft_std": 0.0066154426895082, "reward_judge_quality_mean": 0.2524999976158142, "reward_judge_quality_std": 0.06902380287647247, "reward_total_composite_mean": 0.17487716674804688, "reward_total_composite_std": 0.051213208585977554} {"timestamp_utc": "2026-04-13T12:47:17Z", "mode": "train", "global_step": 14, "epoch": 0.0014063284781516826, "loss": 0.1427, "grad_norm": 8.394230842590332, "learning_rate": 9.960606060606062e-06, "num_tokens": 28140.0, "completions/mean_length": 130.375, "completions/min_length": 90.0, "completions/max_length": 180.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 130.375, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 180.0, "rewards/meter/mean": 0.9129445552825928, "rewards/meter/std": 0.1363978236913681, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9947641491889954, "rewards/repeat_soft/std": 0.008366446942090988, "rewards/judge_quality/mean": 0.273749977350235, "rewards/judge_quality/std": 0.09085899591445923, "rewards/total_composite/mean": 0.2543025612831116, "rewards/total_composite/std": 0.07832667231559753, "reward": 0.2543025612831116, "reward_std": 0.07832667976617813, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21353746950626373, "sampling/sampling_logp_difference/max": 1.6659069061279297, "sampling/importance_sampling_ratio/min": 0.18901915848255157, "sampling/importance_sampling_ratio/mean": 1.0240132808685303, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4317821264266968, "clip_ratio/low_mean": 0.14206395111978054, "clip_ratio/low_min": 0.14206395111978054, "clip_ratio/high_mean": 0.05798611044883728, "clip_ratio/high_max": 0.05798611044883728, "clip_ratio/region_mean": 0.20005006156861782, "reward_total_mean": 0.2543025612831116, "reward_meter_mean": 0.9129445552825928, "reward_meter_std": 0.1363978236913681, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9947641491889954, "reward_repeat_soft_std": 0.008366446942090988, "reward_judge_quality_mean": 0.273749977350235, "reward_judge_quality_std": 0.09085899591445923, "reward_total_composite_mean": 0.2543025612831116, "reward_total_composite_std": 0.07832667231559753} {"timestamp_utc": "2026-04-13T12:47:23Z", "mode": "train", "global_step": 15, "epoch": 0.0015067805123053742, "loss": 0.1828, "grad_norm": 19.198348999023438, "learning_rate": 9.957575757575757e-06, "num_tokens": 29803.0, "completions/mean_length": 42.875, "completions/min_length": 30.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.875, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.6790435314178467, "rewards/meter/std": 0.25500166416168213, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9930763244628906, "rewards/repeat_soft/std": 0.009657313115894794, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.18715444207191467, "rewards/total_composite/mean": 0.30946359038352966, "rewards/total_composite/std": 0.11527153849601746, "reward": 0.30946359038352966, "reward_std": 0.11527153104543686, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2140396237373352, "sampling/sampling_logp_difference/max": 1.9556961059570312, "sampling/importance_sampling_ratio/min": 0.141465961933136, "sampling/importance_sampling_ratio/mean": 1.0327602624893188, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7300214618444443, "clip_ratio/low_mean": 0.13212848082184792, "clip_ratio/low_min": 0.13212848082184792, "clip_ratio/high_mean": 0.08611111342906952, "clip_ratio/high_max": 0.08611111342906952, "clip_ratio/region_mean": 0.21823959425091743, "reward_total_mean": 0.30946359038352966, "reward_meter_mean": 0.6790435314178467, "reward_meter_std": 0.25500166416168213, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9930763244628906, "reward_repeat_soft_std": 0.009657313115894794, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.18715444207191467, "reward_total_composite_mean": 0.30946359038352966, "reward_total_composite_std": 0.11527153849601746} {"timestamp_utc": "2026-04-13T12:47:29Z", "mode": "train", "global_step": 16, "epoch": 0.0016072325464590658, "loss": 0.0641, "grad_norm": 16.974000930786133, "learning_rate": 9.954545454545456e-06, "num_tokens": 31395.0, "completions/mean_length": 48.0, "completions/min_length": 42.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.0, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.4550546109676361, "rewards/meter/std": 0.45469674468040466, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9949187636375427, "rewards/repeat_soft/std": 0.0085137402638793, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.2686706781387329, "rewards/total_composite/mean": 0.25859498977661133, "rewards/total_composite/std": 0.23468804359436035, "reward": 0.25859498977661133, "reward_std": 0.23468804359436035, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2124183177947998, "sampling/sampling_logp_difference/max": 1.2455048561096191, "sampling/importance_sampling_ratio/min": 0.2877956032752991, "sampling/importance_sampling_ratio/mean": 1.040221929550171, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.976992592215538, "clip_ratio/low_mean": 0.1438023690134287, "clip_ratio/low_min": 0.1438023690134287, "clip_ratio/high_mean": 0.06051587499678135, "clip_ratio/high_max": 0.06051587499678135, "clip_ratio/region_mean": 0.20431824401021004, "reward_total_mean": 0.25859498977661133, "reward_meter_mean": 0.4550546109676361, "reward_meter_std": 0.45469674468040466, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9949187636375427, "reward_repeat_soft_std": 0.0085137402638793, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.2686706781387329, "reward_total_composite_mean": 0.25859498977661133, "reward_total_composite_std": 0.23468804359436035} {"timestamp_utc": "2026-04-13T12:47:35Z", "mode": "train", "global_step": 17, "epoch": 0.0017076845806127574, "loss": 0.1538, "grad_norm": 15.405744552612305, "learning_rate": 9.951515151515152e-06, "num_tokens": 33024.0, "completions/mean_length": 44.625, "completions/min_length": 33.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.625, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.49273881316185, "rewards/meter/std": 0.3816235363483429, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9926319718360901, "rewards/repeat_soft/std": 0.012840799987316132, "rewards/judge_quality/mean": 0.3062499761581421, "rewards/judge_quality/std": 0.09500939399003983, "rewards/total_composite/mean": 0.19640883803367615, "rewards/total_composite/std": 0.07236796617507935, "reward": 0.19640883803367615, "reward_std": 0.07236796617507935, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1962367445230484, "sampling/sampling_logp_difference/max": 1.4637451171875, "sampling/importance_sampling_ratio/min": 0.23136815428733826, "sampling/importance_sampling_ratio/mean": 1.0287688970565796, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7023908197879791, "clip_ratio/low_mean": 0.09323069825768471, "clip_ratio/low_min": 0.09323069825768471, "clip_ratio/high_mean": 0.08432095218449831, "clip_ratio/high_max": 0.08432095218449831, "clip_ratio/region_mean": 0.17755165044218302, "reward_total_mean": 0.19640883803367615, "reward_meter_mean": 0.49273881316185, "reward_meter_std": 0.3816235363483429, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9926319718360901, "reward_repeat_soft_std": 0.012840799987316132, "reward_judge_quality_mean": 0.3062499761581421, "reward_judge_quality_std": 0.09500939399003983, "reward_total_composite_mean": 0.19640883803367615, "reward_total_composite_std": 0.07236796617507935} {"timestamp_utc": "2026-04-13T12:47:40Z", "mode": "train", "global_step": 18, "epoch": 0.001808136614766449, "loss": 0.1091, "grad_norm": 19.118051528930664, "learning_rate": 9.948484848484849e-06, "num_tokens": 34458.0, "completions/mean_length": 30.25, "completions/min_length": 20.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.25, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.669553816318512, "rewards/meter/std": 0.4602260887622833, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.27125000953674316, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.20250649750232697, "rewards/total_composite/std": 0.06498918682336807, "reward": 0.20250649750232697, "reward_std": 0.06498918682336807, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21079884469509125, "sampling/sampling_logp_difference/max": 1.7254924774169922, "sampling/importance_sampling_ratio/min": 0.1780853271484375, "sampling/importance_sampling_ratio/mean": 1.0141559839248657, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.862677589058876, "clip_ratio/low_mean": 0.050109206698834896, "clip_ratio/low_min": 0.050109206698834896, "clip_ratio/high_mean": 0.13927721232175827, "clip_ratio/high_max": 0.13927721232175827, "clip_ratio/region_mean": 0.18938641902059317, "reward_total_mean": 0.20250649750232697, "reward_meter_mean": 0.669553816318512, "reward_meter_std": 0.4602260887622833, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.27125000953674316, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.20250649750232697, "reward_total_composite_std": 0.06498918682336807} {"timestamp_utc": "2026-04-13T12:47:47Z", "mode": "train", "global_step": 19, "epoch": 0.0019085886489201406, "loss": 0.0388, "grad_norm": 15.521472930908203, "learning_rate": 9.945454545454546e-06, "num_tokens": 36130.0, "completions/mean_length": 54.0, "completions/min_length": 29.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.0, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.4761258363723755, "rewards/meter/std": 0.4134525954723358, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9986214637756348, "rewards/repeat_soft/std": 0.0037080326583236456, "rewards/judge_quality/mean": 0.3062499761581421, "rewards/judge_quality/std": 0.09500939399003983, "rewards/total_composite/mean": 0.18781903386116028, "rewards/total_composite/std": 0.05385388433933258, "reward": 0.18781903386116028, "reward_std": 0.05385388433933258, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20062369108200073, "sampling/sampling_logp_difference/max": 2.039299964904785, "sampling/importance_sampling_ratio/min": 0.13011977076530457, "sampling/importance_sampling_ratio/mean": 1.008468747138977, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3630243688821793, "clip_ratio/low_mean": 0.07949496805667877, "clip_ratio/low_min": 0.07949496805667877, "clip_ratio/high_mean": 0.10655375942587852, "clip_ratio/high_max": 0.10655375942587852, "clip_ratio/region_mean": 0.1860487274825573, "reward_total_mean": 0.18781903386116028, "reward_meter_mean": 0.4761258363723755, "reward_meter_std": 0.4134525954723358, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9986214637756348, "reward_repeat_soft_std": 0.0037080326583236456, "reward_judge_quality_mean": 0.3062499761581421, "reward_judge_quality_std": 0.09500939399003983, "reward_total_composite_mean": 0.18781903386116028, "reward_total_composite_std": 0.05385388433933258} {"timestamp_utc": "2026-04-13T12:47:53Z", "mode": "train", "global_step": 20, "epoch": 0.0020090406830738324, "loss": 0.0749, "grad_norm": 10.678475379943848, "learning_rate": 9.942424242424244e-06, "num_tokens": 38259.0, "completions/mean_length": 79.125, "completions/min_length": 59.0, "completions/max_length": 103.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.125, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 103.0, "rewards/meter/mean": 0.6724998950958252, "rewards/meter/std": 0.3194420635700226, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9975677728652954, "rewards/repeat_soft/std": 0.003126713912934065, "rewards/judge_quality/mean": 0.2562499940395355, "rewards/judge_quality/std": 0.06781013309955597, "rewards/total_composite/mean": 0.20345285534858704, "rewards/total_composite/std": 0.09404149651527405, "reward": 0.20345285534858704, "reward_std": 0.09404149651527405, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.25104182958602905, "sampling/sampling_logp_difference/max": 1.7388010025024414, "sampling/importance_sampling_ratio/min": 0.17573098838329315, "sampling/importance_sampling_ratio/mean": 1.0093882083892822, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2550461143255234, "clip_ratio/low_mean": 0.18275271356105804, "clip_ratio/low_min": 0.18275271356105804, "clip_ratio/high_mean": 0.06071428768336773, "clip_ratio/high_max": 0.06071428768336773, "clip_ratio/region_mean": 0.24346700124442577, "reward_total_mean": 0.20345285534858704, "reward_meter_mean": 0.6724998950958252, "reward_meter_std": 0.3194420635700226, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9975677728652954, "reward_repeat_soft_std": 0.003126713912934065, "reward_judge_quality_mean": 0.2562499940395355, "reward_judge_quality_std": 0.06781013309955597, "reward_total_composite_mean": 0.20345285534858704, "reward_total_composite_std": 0.09404149651527405} {"timestamp_utc": "2026-04-13T12:47:59Z", "mode": "train", "global_step": 21, "epoch": 0.002109492717227524, "loss": 0.1033, "grad_norm": 28.266448974609375, "learning_rate": 9.939393939393939e-06, "num_tokens": 39787.0, "completions/mean_length": 32.0, "completions/min_length": 21.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.45389506220817566, "rewards/meter/std": 0.39092472195625305, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9942681193351746, "rewards/repeat_soft/std": 0.006455856841057539, "rewards/judge_quality/mean": 0.3812499940395355, "rewards/judge_quality/std": 0.166770800948143, "rewards/total_composite/mean": 0.26310059428215027, "rewards/total_composite/std": 0.18096305429935455, "reward": 0.26310059428215027, "reward_std": 0.18096303939819336, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21730417013168335, "sampling/sampling_logp_difference/max": 1.5662403106689453, "sampling/importance_sampling_ratio/min": 0.20882883667945862, "sampling/importance_sampling_ratio/mean": 1.010674238204956, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1850305646657944, "clip_ratio/low_mean": 0.08216887293383479, "clip_ratio/low_min": 0.08216887293383479, "clip_ratio/high_mean": 0.07137393951416016, "clip_ratio/high_max": 0.07137393951416016, "clip_ratio/region_mean": 0.15354281244799495, "reward_total_mean": 0.26310059428215027, "reward_meter_mean": 0.45389506220817566, "reward_meter_std": 0.39092472195625305, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9942681193351746, "reward_repeat_soft_std": 0.006455856841057539, "reward_judge_quality_mean": 0.3812499940395355, "reward_judge_quality_std": 0.166770800948143, "reward_total_composite_mean": 0.26310059428215027, "reward_total_composite_std": 0.18096305429935455} {"timestamp_utc": "2026-04-13T12:48:05Z", "mode": "train", "global_step": 22, "epoch": 0.0022099447513812156, "loss": 0.0779, "grad_norm": 14.22166633605957, "learning_rate": 9.936363636363638e-06, "num_tokens": 41507.0, "completions/mean_length": 56.0, "completions/min_length": 39.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.0, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9883540868759155, "rewards/meter/std": 0.009184768423438072, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9959346055984497, "rewards/repeat_soft/std": 0.005422628950327635, "rewards/judge_quality/mean": 0.23875001072883606, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.2367658019065857, "rewards/total_composite/std": 0.014922114089131355, "reward": 0.2367658019065857, "reward_std": 0.014922115951776505, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20160622894763947, "sampling/sampling_logp_difference/max": 1.526987075805664, "sampling/importance_sampling_ratio/min": 0.26236656308174133, "sampling/importance_sampling_ratio/mean": 1.0414557456970215, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.040539041161537, "clip_ratio/low_mean": 0.07369768805801868, "clip_ratio/low_min": 0.07369768805801868, "clip_ratio/high_mean": 0.11720447614789009, "clip_ratio/high_max": 0.11720447614789009, "clip_ratio/region_mean": 0.19090216420590878, "reward_total_mean": 0.2367658019065857, "reward_meter_mean": 0.9883540868759155, "reward_meter_std": 0.009184768423438072, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9959346055984497, "reward_repeat_soft_std": 0.005422628950327635, "reward_judge_quality_mean": 0.23875001072883606, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.2367658019065857, "reward_total_composite_std": 0.014922114089131355} {"timestamp_utc": "2026-04-13T12:48:11Z", "mode": "train", "global_step": 23, "epoch": 0.0023103967855349072, "loss": 0.0537, "grad_norm": 16.252086639404297, "learning_rate": 9.933333333333334e-06, "num_tokens": 43114.0, "completions/mean_length": 40.875, "completions/min_length": 36.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.875, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.6953567266464233, "rewards/meter/std": 0.3707497715950012, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9990596175193787, "rewards/repeat_soft/std": 0.0017459453083574772, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.2291552722454071, "rewards/total_composite/std": 0.12831708788871765, "reward": 0.2291552722454071, "reward_std": 0.12831710278987885, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21670785546302795, "sampling/sampling_logp_difference/max": 1.927046775817871, "sampling/importance_sampling_ratio/min": 0.14557749032974243, "sampling/importance_sampling_ratio/mean": 1.029430866241455, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7587376534938812, "clip_ratio/low_mean": 0.09989950433373451, "clip_ratio/low_min": 0.09989950433373451, "clip_ratio/high_mean": 0.10116968676447868, "clip_ratio/high_max": 0.10116968676447868, "clip_ratio/region_mean": 0.2010691910982132, "reward_total_mean": 0.2291552722454071, "reward_meter_mean": 0.6953567266464233, "reward_meter_std": 0.3707497715950012, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9990596175193787, "reward_repeat_soft_std": 0.0017459453083574772, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.2291552722454071, "reward_total_composite_std": 0.12831708788871765} {"timestamp_utc": "2026-04-13T12:48:17Z", "mode": "train", "global_step": 24, "epoch": 0.002410848819688599, "loss": 0.0181, "grad_norm": 19.029216766357422, "learning_rate": 9.930303030303031e-06, "num_tokens": 44808.0, "completions/mean_length": 59.75, "completions/min_length": 46.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.75, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.45434120297431946, "rewards/meter/std": 0.4283906817436218, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9990161657333374, "rewards/repeat_soft/std": 0.002018744358792901, "rewards/judge_quality/mean": 0.3062500059604645, "rewards/judge_quality/std": 0.2081165611743927, "rewards/total_composite/mean": 0.22358007729053497, "rewards/total_composite/std": 0.2376146912574768, "reward": 0.22358007729053497, "reward_std": 0.2376146912574768, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.213044673204422, "sampling/sampling_logp_difference/max": 1.6430997848510742, "sampling/importance_sampling_ratio/min": 0.19337967038154602, "sampling/importance_sampling_ratio/mean": 1.0326234102249146, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5965741500258446, "clip_ratio/low_mean": 0.1360377911478281, "clip_ratio/low_min": 0.1360377911478281, "clip_ratio/high_mean": 0.05652702413499355, "clip_ratio/high_max": 0.05652702413499355, "clip_ratio/region_mean": 0.19256481528282166, "reward_total_mean": 0.22358007729053497, "reward_meter_mean": 0.45434120297431946, "reward_meter_std": 0.4283906817436218, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9990161657333374, "reward_repeat_soft_std": 0.002018744358792901, "reward_judge_quality_mean": 0.3062500059604645, "reward_judge_quality_std": 0.2081165611743927, "reward_total_composite_mean": 0.22358007729053497, "reward_total_composite_std": 0.2376146912574768} {"timestamp_utc": "2026-04-13T12:48:25Z", "mode": "train", "global_step": 25, "epoch": 0.0025113008538422904, "loss": 0.0051, "grad_norm": 7.79266881942749, "learning_rate": 9.927272727272728e-06, "num_tokens": 47482.0, "completions/mean_length": 125.25, "completions/min_length": 110.0, "completions/max_length": 142.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 125.25, "completions/min_terminated_length": 110.0, "completions/max_terminated_length": 142.0, "rewards/meter/mean": 0.38681113719940186, "rewards/meter/std": 0.3000413477420807, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9985337257385254, "rewards/repeat_soft/std": 0.000780539121478796, "rewards/judge_quality/mean": 0.3387500047683716, "rewards/judge_quality/std": 0.20538726449012756, "rewards/total_composite/mean": 0.20475128293037415, "rewards/total_composite/std": 0.1320890635251999, "reward": 0.20475128293037415, "reward_std": 0.1320890486240387, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16238392889499664, "sampling/sampling_logp_difference/max": 1.6394009590148926, "sampling/importance_sampling_ratio/min": 0.19409628212451935, "sampling/importance_sampling_ratio/mean": 1.0219614505767822, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.1800087988376617, "clip_ratio/low_mean": 0.09639591071754694, "clip_ratio/low_min": 0.09639591071754694, "clip_ratio/high_mean": 0.04710742551833391, "clip_ratio/high_max": 0.04710742551833391, "clip_ratio/region_mean": 0.14350333623588085, "reward_total_mean": 0.20475128293037415, "reward_meter_mean": 0.38681113719940186, "reward_meter_std": 0.3000413477420807, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9985337257385254, "reward_repeat_soft_std": 0.000780539121478796, "reward_judge_quality_mean": 0.3387500047683716, "reward_judge_quality_std": 0.20538726449012756, "reward_total_composite_mean": 0.20475128293037415, "reward_total_composite_std": 0.1320890635251999} {"timestamp_utc": "2026-04-13T12:48:31Z", "mode": "train", "global_step": 26, "epoch": 0.002611752887995982, "loss": -0.1191, "grad_norm": 17.39537811279297, "learning_rate": 9.924242424242425e-06, "num_tokens": 48947.0, "completions/mean_length": 26.125, "completions/min_length": 17.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.125, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9929069876670837, "rewards/meter/std": 0.011440408416092396, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9495017528533936, "rewards/repeat_soft/std": 0.024305421859025955, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.393950879573822, "rewards/total_composite/std": 0.05983329936861992, "reward": 0.393950879573822, "reward_std": 0.05983330309391022, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0914190411567688, "sampling/sampling_logp_difference/max": 1.7267370223999023, "sampling/importance_sampling_ratio/min": 0.17786382138729095, "sampling/importance_sampling_ratio/mean": 1.0033172369003296, "sampling/importance_sampling_ratio/max": 1.9017900228500366, "entropy": 0.5844643115997314, "clip_ratio/low_mean": 0.029411764815449715, "clip_ratio/low_min": 0.029411764815449715, "clip_ratio/high_mean": 0.0750319343060255, "clip_ratio/high_max": 0.0750319343060255, "clip_ratio/region_mean": 0.10444369912147522, "reward_total_mean": 0.393950879573822, "reward_meter_mean": 0.9929069876670837, "reward_meter_std": 0.011440408416092396, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9495017528533936, "reward_repeat_soft_std": 0.024305421859025955, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.393950879573822, "reward_total_composite_std": 0.05983329936861992} {"timestamp_utc": "2026-04-13T12:48:38Z", "mode": "train", "global_step": 27, "epoch": 0.0027122049221496736, "loss": 0.0134, "grad_norm": 10.1174898147583, "learning_rate": 9.921212121212121e-06, "num_tokens": 51538.0, "completions/mean_length": 126.875, "completions/min_length": 96.0, "completions/max_length": 174.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 126.875, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 174.0, "rewards/meter/mean": 0.5053344368934631, "rewards/meter/std": 0.41090095043182373, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9992033839225769, "rewards/repeat_soft/std": 0.0007164619164541364, "rewards/judge_quality/mean": 0.29499998688697815, "rewards/judge_quality/std": 0.1035098284482956, "rewards/total_composite/mean": 0.20223358273506165, "rewards/total_composite/std": 0.10832107067108154, "reward": 0.20223358273506165, "reward_std": 0.10832107812166214, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2293359339237213, "sampling/sampling_logp_difference/max": 1.6479358673095703, "sampling/importance_sampling_ratio/min": 0.1924467384815216, "sampling/importance_sampling_ratio/mean": 1.0598264932632446, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6311583518981934, "clip_ratio/low_mean": 0.07886148244142532, "clip_ratio/low_min": 0.07886148244142532, "clip_ratio/high_mean": 0.13825739920139313, "clip_ratio/high_max": 0.13825739920139313, "clip_ratio/region_mean": 0.21711888164281845, "reward_total_mean": 0.20223358273506165, "reward_meter_mean": 0.5053344368934631, "reward_meter_std": 0.41090095043182373, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9992033839225769, "reward_repeat_soft_std": 0.0007164619164541364, "reward_judge_quality_mean": 0.29499998688697815, "reward_judge_quality_std": 0.1035098284482956, "reward_total_composite_mean": 0.20223358273506165, "reward_total_composite_std": 0.10832107067108154} {"timestamp_utc": "2026-04-13T12:48:44Z", "mode": "train", "global_step": 28, "epoch": 0.0028126569563033652, "loss": -0.0288, "grad_norm": 17.79926872253418, "learning_rate": 9.918181818181818e-06, "num_tokens": 53101.0, "completions/mean_length": 33.375, "completions/min_length": 18.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.375, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.759634256362915, "rewards/meter/std": 0.4249105453491211, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.94775390625, "rewards/repeat_soft/std": 0.041864264756441116, "rewards/judge_quality/mean": 0.3374999761581421, "rewards/judge_quality/std": 0.21796134114265442, "rewards/total_composite/mean": 0.2884432077407837, "rewards/total_composite/std": 0.24120306968688965, "reward": 0.2884432077407837, "reward_std": 0.24120306968688965, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1766754537820816, "sampling/sampling_logp_difference/max": 1.2298507690429688, "sampling/importance_sampling_ratio/min": 0.29233619570732117, "sampling/importance_sampling_ratio/mean": 1.024755835533142, "sampling/importance_sampling_ratio/max": 1.9423034191131592, "entropy": 1.6584673970937729, "clip_ratio/low_mean": 0.14047385100275278, "clip_ratio/low_min": 0.14047385100275278, "clip_ratio/high_mean": 0.03227124270051718, "clip_ratio/high_max": 0.03227124270051718, "clip_ratio/region_mean": 0.17274509370326996, "reward_total_mean": 0.2884432077407837, "reward_meter_mean": 0.759634256362915, "reward_meter_std": 0.4249105453491211, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.94775390625, "reward_repeat_soft_std": 0.041864264756441116, "reward_judge_quality_mean": 0.3374999761581421, "reward_judge_quality_std": 0.21796134114265442, "reward_total_composite_mean": 0.2884432077407837, "reward_total_composite_std": 0.24120306968688965} {"timestamp_utc": "2026-04-13T12:48:52Z", "mode": "train", "global_step": 29, "epoch": 0.002913108990457057, "loss": 0.0726, "grad_norm": 7.879302978515625, "learning_rate": 9.915151515151515e-06, "num_tokens": 56052.0, "completions/mean_length": 168.875, "completions/min_length": 140.0, "completions/max_length": 195.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 168.875, "completions/min_terminated_length": 140.0, "completions/max_terminated_length": 195.0, "rewards/meter/mean": 0.6800090074539185, "rewards/meter/std": 0.3618488311767578, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9975203275680542, "rewards/repeat_soft/std": 0.0027590380050241947, "rewards/judge_quality/mean": 0.2150000035762787, "rewards/judge_quality/std": 0.014142133295536041, "rewards/total_composite/mean": 0.15685734152793884, "rewards/total_composite/std": 0.08051047474145889, "reward": 0.15685734152793884, "reward_std": 0.08051047474145889, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2022891640663147, "sampling/sampling_logp_difference/max": 1.8225994110107422, "sampling/importance_sampling_ratio/min": 0.17849254608154297, "sampling/importance_sampling_ratio/mean": 1.0238168239593506, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2702658623456955, "clip_ratio/low_mean": 0.06336063705384731, "clip_ratio/low_min": 0.06336063705384731, "clip_ratio/high_mean": 0.13224178925156593, "clip_ratio/high_max": 0.13224178925156593, "clip_ratio/region_mean": 0.19560242630541325, "reward_total_mean": 0.15685734152793884, "reward_meter_mean": 0.6800090074539185, "reward_meter_std": 0.3618488311767578, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9975203275680542, "reward_repeat_soft_std": 0.0027590380050241947, "reward_judge_quality_mean": 0.2150000035762787, "reward_judge_quality_std": 0.014142133295536041, "reward_total_composite_mean": 0.15685734152793884, "reward_total_composite_std": 0.08051047474145889} {"timestamp_utc": "2026-04-13T12:48:58Z", "mode": "train", "global_step": 30, "epoch": 0.0030135610246107484, "loss": -0.0567, "grad_norm": 17.424236297607422, "learning_rate": 9.912121212121213e-06, "num_tokens": 57912.0, "completions/mean_length": 48.5, "completions/min_length": 34.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.5, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.6463769674301147, "rewards/meter/std": 0.40498650074005127, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.994937539100647, "rewards/repeat_soft/std": 0.006559374276548624, "rewards/judge_quality/mean": 0.26999998092651367, "rewards/judge_quality/std": 0.1414213627576828, "rewards/total_composite/mean": 0.19401989877223969, "rewards/total_composite/std": 0.06956040859222412, "reward": 0.19401989877223969, "reward_std": 0.06956040114164352, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22575657069683075, "sampling/sampling_logp_difference/max": 1.6284923553466797, "sampling/importance_sampling_ratio/min": 0.19622518122196198, "sampling/importance_sampling_ratio/mean": 1.0097166299819946, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3254844695329666, "clip_ratio/low_mean": 0.060775401070714, "clip_ratio/low_min": 0.060775401070714, "clip_ratio/high_mean": 0.14530975557863712, "clip_ratio/high_max": 0.14530975557863712, "clip_ratio/region_mean": 0.20608515664935112, "reward_total_mean": 0.19401989877223969, "reward_meter_mean": 0.6463769674301147, "reward_meter_std": 0.40498650074005127, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.994937539100647, "reward_repeat_soft_std": 0.006559374276548624, "reward_judge_quality_mean": 0.26999998092651367, "reward_judge_quality_std": 0.1414213627576828, "reward_total_composite_mean": 0.19401989877223969, "reward_total_composite_std": 0.06956040859222412} {"timestamp_utc": "2026-04-13T12:49:04Z", "mode": "train", "global_step": 31, "epoch": 0.00311401305876444, "loss": 0.1641, "grad_norm": 16.771867752075195, "learning_rate": 9.90909090909091e-06, "num_tokens": 59861.0, "completions/mean_length": 71.625, "completions/min_length": 47.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.625, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.23064470291137695, "rewards/meter/std": 0.14039665460586548, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9896365404129028, "rewards/repeat_soft/std": 0.011847006157040596, "rewards/judge_quality/mean": 0.3449999988079071, "rewards/judge_quality/std": 0.14880476891994476, "rewards/total_composite/mean": 0.172789067029953, "rewards/total_composite/std": 0.09928283095359802, "reward": 0.172789067029953, "reward_std": 0.09928283095359802, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2790986895561218, "sampling/sampling_logp_difference/max": 3.0018956661224365, "sampling/importance_sampling_ratio/min": 0.0874783918261528, "sampling/importance_sampling_ratio/mean": 1.0112502574920654, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4137107655405998, "clip_ratio/low_mean": 0.1759606134146452, "clip_ratio/low_min": 0.1759606134146452, "clip_ratio/high_mean": 0.06050531938672066, "clip_ratio/high_max": 0.06050531938672066, "clip_ratio/region_mean": 0.23646593280136585, "reward_total_mean": 0.172789067029953, "reward_meter_mean": 0.23064470291137695, "reward_meter_std": 0.14039665460586548, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9896365404129028, "reward_repeat_soft_std": 0.011847006157040596, "reward_judge_quality_mean": 0.3449999988079071, "reward_judge_quality_std": 0.14880476891994476, "reward_total_composite_mean": 0.172789067029953, "reward_total_composite_std": 0.09928283095359802} {"timestamp_utc": "2026-04-13T12:49:10Z", "mode": "train", "global_step": 32, "epoch": 0.0032144650929181316, "loss": 0.0739, "grad_norm": 26.786174774169922, "learning_rate": 9.906060606060607e-06, "num_tokens": 61299.0, "completions/mean_length": 24.75, "completions/min_length": 16.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.75, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.29178377985954285, "rewards/meter/std": 0.3586907386779785, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.961837112903595, "rewards/repeat_soft/std": 0.0018749026348814368, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.19486258924007416, "rewards/total_composite/mean": 0.23482491075992584, "rewards/total_composite/std": 0.2344757467508316, "reward": 0.23482491075992584, "reward_std": 0.2344757318496704, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23777849972248077, "sampling/sampling_logp_difference/max": 2.170952320098877, "sampling/importance_sampling_ratio/min": 0.11406894028186798, "sampling/importance_sampling_ratio/mean": 1.0599104166030884, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4571485072374344, "clip_ratio/low_mean": 0.18879137095063925, "clip_ratio/low_min": 0.18879137095063925, "clip_ratio/high_mean": 0.04714912362396717, "clip_ratio/high_max": 0.04714912362396717, "clip_ratio/region_mean": 0.23594049457460642, "reward_total_mean": 0.23482491075992584, "reward_meter_mean": 0.29178377985954285, "reward_meter_std": 0.3586907386779785, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.961837112903595, "reward_repeat_soft_std": 0.0018749026348814368, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.19486258924007416, "reward_total_composite_mean": 0.23482491075992584, "reward_total_composite_std": 0.2344757467508316} {"timestamp_utc": "2026-04-13T12:49:16Z", "mode": "train", "global_step": 33, "epoch": 0.0033149171270718232, "loss": 0.036, "grad_norm": 17.3479061126709, "learning_rate": 9.903030303030305e-06, "num_tokens": 62947.0, "completions/mean_length": 42.0, "completions/min_length": 34.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.0, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.6314820051193237, "rewards/meter/std": 0.4792705774307251, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9985815286636353, "rewards/repeat_soft/std": 0.003543524071574211, "rewards/judge_quality/mean": 0.2562499940395355, "rewards/judge_quality/std": 0.06781013309955597, "rewards/total_composite/mean": 0.20051582157611847, "rewards/total_composite/std": 0.11133254319429398, "reward": 0.20051582157611847, "reward_std": 0.11133254319429398, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20595410466194153, "sampling/sampling_logp_difference/max": 1.4103212356567383, "sampling/importance_sampling_ratio/min": 0.24406488239765167, "sampling/importance_sampling_ratio/mean": 1.0241533517837524, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.233390688896179, "clip_ratio/low_mean": 0.046910145319998264, "clip_ratio/low_min": 0.046910145319998264, "clip_ratio/high_mean": 0.13978325575590134, "clip_ratio/high_max": 0.13978325575590134, "clip_ratio/region_mean": 0.1866934010758996, "reward_total_mean": 0.20051582157611847, "reward_meter_mean": 0.6314820051193237, "reward_meter_std": 0.4792705774307251, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9985815286636353, "reward_repeat_soft_std": 0.003543524071574211, "reward_judge_quality_mean": 0.2562499940395355, "reward_judge_quality_std": 0.06781013309955597, "reward_total_composite_mean": 0.20051582157611847, "reward_total_composite_std": 0.11133254319429398} {"timestamp_utc": "2026-04-13T12:49:22Z", "mode": "train", "global_step": 34, "epoch": 0.003415369161225515, "loss": 0.0314, "grad_norm": 33.14555740356445, "learning_rate": 9.9e-06, "num_tokens": 64491.0, "completions/mean_length": 35.0, "completions/min_length": 28.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.0, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.28801655769348145, "rewards/meter/std": 0.3693912923336029, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9700682163238525, "rewards/repeat_soft/std": 0.0731503814458847, "rewards/judge_quality/mean": 0.2849999964237213, "rewards/judge_quality/std": 0.0843462198972702, "rewards/total_composite/mean": 0.14035621285438538, "rewards/total_composite/std": 0.055986013263463974, "reward": 0.14035621285438538, "reward_std": 0.055986013263463974, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23407866060733795, "sampling/sampling_logp_difference/max": 1.9833375215530396, "sampling/importance_sampling_ratio/min": 0.13760919868946075, "sampling/importance_sampling_ratio/mean": 1.0082929134368896, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.644989162683487, "clip_ratio/low_mean": 0.0862089991569519, "clip_ratio/low_min": 0.0862089991569519, "clip_ratio/high_mean": 0.14916404150426388, "clip_ratio/high_max": 0.14916404150426388, "clip_ratio/region_mean": 0.23537304066121578, "reward_total_mean": 0.14035621285438538, "reward_meter_mean": 0.28801655769348145, "reward_meter_std": 0.3693912923336029, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9700682163238525, "reward_repeat_soft_std": 0.0731503814458847, "reward_judge_quality_mean": 0.2849999964237213, "reward_judge_quality_std": 0.0843462198972702, "reward_total_composite_mean": 0.14035621285438538, "reward_total_composite_std": 0.055986013263463974} {"timestamp_utc": "2026-04-13T12:49:28Z", "mode": "train", "global_step": 35, "epoch": 0.0035158211953792064, "loss": 0.0637, "grad_norm": 19.889467239379883, "learning_rate": 9.896969696969699e-06, "num_tokens": 66075.0, "completions/mean_length": 33.0, "completions/min_length": 27.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.0, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.4673800468444824, "rewards/meter/std": 0.4234526753425598, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9926177263259888, "rewards/repeat_soft/std": 0.006717136595398188, "rewards/judge_quality/mean": 0.45625001192092896, "rewards/judge_quality/std": 0.18368743360042572, "rewards/total_composite/mean": 0.28530552983283997, "rewards/total_composite/std": 0.2261982560157776, "reward": 0.28530552983283997, "reward_std": 0.2261982560157776, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20924991369247437, "sampling/sampling_logp_difference/max": 3.056974411010742, "sampling/importance_sampling_ratio/min": 0.0470297709107399, "sampling/importance_sampling_ratio/mean": 0.999767541885376, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.8244521431624889, "clip_ratio/low_mean": 0.1276165684685111, "clip_ratio/low_min": 0.1276165684685111, "clip_ratio/high_mean": 0.02765804622322321, "clip_ratio/high_max": 0.02765804622322321, "clip_ratio/region_mean": 0.15527461469173431, "reward_total_mean": 0.28530552983283997, "reward_meter_mean": 0.4673800468444824, "reward_meter_std": 0.4234526753425598, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9926177263259888, "reward_repeat_soft_std": 0.006717136595398188, "reward_judge_quality_mean": 0.45625001192092896, "reward_judge_quality_std": 0.18368743360042572, "reward_total_composite_mean": 0.28530552983283997, "reward_total_composite_std": 0.2261982560157776} {"timestamp_utc": "2026-04-13T12:49:34Z", "mode": "train", "global_step": 36, "epoch": 0.003616273229532898, "loss": 0.0973, "grad_norm": 16.561141967773438, "learning_rate": 9.893939393939395e-06, "num_tokens": 67614.0, "completions/mean_length": 43.375, "completions/min_length": 31.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.375, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.4614804983139038, "rewards/meter/std": 0.4654310643672943, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9922767877578735, "rewards/repeat_soft/std": 0.0168900229036808, "rewards/judge_quality/mean": 0.38124996423721313, "rewards/judge_quality/std": 0.12799972295761108, "rewards/total_composite/mean": 0.2614460587501526, "rewards/total_composite/std": 0.18071474134922028, "reward": 0.2614460587501526, "reward_std": 0.18071474134922028, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1762866973876953, "sampling/sampling_logp_difference/max": 2.2372875213623047, "sampling/importance_sampling_ratio/min": 0.10674765706062317, "sampling/importance_sampling_ratio/mean": 1.0038046836853027, "sampling/importance_sampling_ratio/max": 1.855939269065857, "entropy": 1.5117772668600082, "clip_ratio/low_mean": 0.08194054756313562, "clip_ratio/low_min": 0.08194054756313562, "clip_ratio/high_mean": 0.051636514253914356, "clip_ratio/high_max": 0.051636514253914356, "clip_ratio/region_mean": 0.13357706181704998, "reward_total_mean": 0.2614460587501526, "reward_meter_mean": 0.4614804983139038, "reward_meter_std": 0.4654310643672943, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9922767877578735, "reward_repeat_soft_std": 0.0168900229036808, "reward_judge_quality_mean": 0.38124996423721313, "reward_judge_quality_std": 0.12799972295761108, "reward_total_composite_mean": 0.2614460587501526, "reward_total_composite_std": 0.18071474134922028} {"timestamp_utc": "2026-04-13T12:49:40Z", "mode": "train", "global_step": 37, "epoch": 0.0037167252636865896, "loss": 0.0809, "grad_norm": 18.028430938720703, "learning_rate": 9.890909090909092e-06, "num_tokens": 69309.0, "completions/mean_length": 40.875, "completions/min_length": 34.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.875, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.39811915159225464, "rewards/meter/std": 0.47292259335517883, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.995173454284668, "rewards/repeat_soft/std": 0.01254533976316452, "rewards/judge_quality/mean": 0.32749998569488525, "rewards/judge_quality/std": 0.14606750011444092, "rewards/total_composite/mean": 0.19178056716918945, "rewards/total_composite/std": 0.11140396445989609, "reward": 0.19178056716918945, "reward_std": 0.11140395700931549, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24725963175296783, "sampling/sampling_logp_difference/max": 1.543391227722168, "sampling/importance_sampling_ratio/min": 0.21365532279014587, "sampling/importance_sampling_ratio/mean": 1.062158226966858, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.5865648090839386, "clip_ratio/low_mean": 0.14958483539521694, "clip_ratio/low_min": 0.14958483539521694, "clip_ratio/high_mean": 0.1176363117992878, "clip_ratio/high_max": 0.1176363117992878, "clip_ratio/region_mean": 0.26722114719450474, "reward_total_mean": 0.19178056716918945, "reward_meter_mean": 0.39811915159225464, "reward_meter_std": 0.47292259335517883, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.995173454284668, "reward_repeat_soft_std": 0.01254533976316452, "reward_judge_quality_mean": 0.32749998569488525, "reward_judge_quality_std": 0.14606750011444092, "reward_total_composite_mean": 0.19178056716918945, "reward_total_composite_std": 0.11140396445989609} {"timestamp_utc": "2026-04-13T12:49:46Z", "mode": "train", "global_step": 38, "epoch": 0.0038171772978402812, "loss": -0.0289, "grad_norm": 11.82272720336914, "learning_rate": 9.887878787878789e-06, "num_tokens": 70984.0, "completions/mean_length": 56.375, "completions/min_length": 40.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.375, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.940577507019043, "rewards/meter/std": 0.0998968780040741, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9988288879394531, "rewards/repeat_soft/std": 0.0020019880030304193, "rewards/judge_quality/mean": 0.32999998331069946, "rewards/judge_quality/std": 0.1928730010986328, "rewards/total_composite/mean": 0.3193818926811218, "rewards/total_composite/std": 0.1936800330877304, "reward": 0.3193818926811218, "reward_std": 0.1936800181865692, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20165720582008362, "sampling/sampling_logp_difference/max": 2.2060861587524414, "sampling/importance_sampling_ratio/min": 0.11013083904981613, "sampling/importance_sampling_ratio/mean": 1.0116405487060547, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7450886070728302, "clip_ratio/low_mean": 0.1500273859128356, "clip_ratio/low_min": 0.1500273859128356, "clip_ratio/high_mean": 0.032031250186264515, "clip_ratio/high_max": 0.032031250186264515, "clip_ratio/region_mean": 0.1820586360991001, "reward_total_mean": 0.3193818926811218, "reward_meter_mean": 0.940577507019043, "reward_meter_std": 0.0998968780040741, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9988288879394531, "reward_repeat_soft_std": 0.0020019880030304193, "reward_judge_quality_mean": 0.32999998331069946, "reward_judge_quality_std": 0.1928730010986328, "reward_total_composite_mean": 0.3193818926811218, "reward_total_composite_std": 0.1936800330877304} {"timestamp_utc": "2026-04-13T12:49:52Z", "mode": "train", "global_step": 39, "epoch": 0.003917629331993973, "loss": 0.0854, "grad_norm": 21.60627555847168, "learning_rate": 9.884848484848486e-06, "num_tokens": 72664.0, "completions/mean_length": 52.0, "completions/min_length": 46.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.0, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.8593474626541138, "rewards/meter/std": 0.059598565101623535, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9931809902191162, "rewards/repeat_soft/std": 0.006509596481919289, "rewards/judge_quality/mean": 0.4737499952316284, "rewards/judge_quality/std": 0.1340509057044983, "rewards/total_composite/mean": 0.4315245747566223, "rewards/total_composite/std": 0.1271274983882904, "reward": 0.4315245747566223, "reward_std": 0.1271274983882904, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16323387622833252, "sampling/sampling_logp_difference/max": 3.1816110610961914, "sampling/importance_sampling_ratio/min": 0.04151871055364609, "sampling/importance_sampling_ratio/mean": 0.991568922996521, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.591550350189209, "clip_ratio/low_mean": 0.09737164154648781, "clip_ratio/low_min": 0.09737164154648781, "clip_ratio/high_mean": 0.025629001669585705, "clip_ratio/high_max": 0.025629001669585705, "clip_ratio/region_mean": 0.12300064321607351, "reward_total_mean": 0.4315245747566223, "reward_meter_mean": 0.8593474626541138, "reward_meter_std": 0.059598565101623535, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9931809902191162, "reward_repeat_soft_std": 0.006509596481919289, "reward_judge_quality_mean": 0.4737499952316284, "reward_judge_quality_std": 0.1340509057044983, "reward_total_composite_mean": 0.4315245747566223, "reward_total_composite_std": 0.1271274983882904} {"timestamp_utc": "2026-04-13T12:49:59Z", "mode": "train", "global_step": 40, "epoch": 0.004018081366147665, "loss": 0.0882, "grad_norm": 16.42487144470215, "learning_rate": 9.881818181818182e-06, "num_tokens": 74308.0, "completions/mean_length": 42.5, "completions/min_length": 30.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.5, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.5914168357849121, "rewards/meter/std": 0.41509711742401123, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9794532060623169, "rewards/repeat_soft/std": 0.03394440561532974, "rewards/judge_quality/mean": 0.5950000286102295, "rewards/judge_quality/std": 0.19820624589920044, "rewards/total_composite/mean": 0.4732854962348938, "rewards/total_composite/std": 0.3093723654747009, "reward": 0.4732854962348938, "reward_std": 0.3093723952770233, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15868167579174042, "sampling/sampling_logp_difference/max": 1.9446430206298828, "sampling/importance_sampling_ratio/min": 0.14303827285766602, "sampling/importance_sampling_ratio/mean": 1.0041155815124512, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0093816071748734, "clip_ratio/low_mean": 0.07824461627751589, "clip_ratio/low_min": 0.07824461627751589, "clip_ratio/high_mean": 0.04752552509307861, "clip_ratio/high_max": 0.04752552509307861, "clip_ratio/region_mean": 0.1257701413705945, "reward_total_mean": 0.4732854962348938, "reward_meter_mean": 0.5914168357849121, "reward_meter_std": 0.41509711742401123, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9794532060623169, "reward_repeat_soft_std": 0.03394440561532974, "reward_judge_quality_mean": 0.5950000286102295, "reward_judge_quality_std": 0.19820624589920044, "reward_total_composite_mean": 0.4732854962348938, "reward_total_composite_std": 0.3093723654747009} {"timestamp_utc": "2026-04-13T12:50:05Z", "mode": "train", "global_step": 41, "epoch": 0.004118533400301356, "loss": 0.1522, "grad_norm": 14.568224906921387, "learning_rate": 9.87878787878788e-06, "num_tokens": 75939.0, "completions/mean_length": 43.875, "completions/min_length": 30.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.875, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.6071714162826538, "rewards/meter/std": 0.39920130372047424, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9951162338256836, "rewards/repeat_soft/std": 0.00768527016043663, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.21553505957126617, "rewards/total_composite/mean": 0.2829132378101349, "rewards/total_composite/std": 0.15281237661838531, "reward": 0.2829132378101349, "reward_std": 0.15281237661838531, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2150118201971054, "sampling/sampling_logp_difference/max": 1.5621118545532227, "sampling/importance_sampling_ratio/min": 0.20969274640083313, "sampling/importance_sampling_ratio/mean": 1.028202772140503, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.5838479697704315, "clip_ratio/low_mean": 0.10096247587352991, "clip_ratio/low_min": 0.10096247587352991, "clip_ratio/high_mean": 0.07940251752734184, "clip_ratio/high_max": 0.07940251752734184, "clip_ratio/region_mean": 0.18036499340087175, "reward_total_mean": 0.2829132378101349, "reward_meter_mean": 0.6071714162826538, "reward_meter_std": 0.39920130372047424, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9951162338256836, "reward_repeat_soft_std": 0.00768527016043663, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.21553505957126617, "reward_total_composite_mean": 0.2829132378101349, "reward_total_composite_std": 0.15281237661838531} {"timestamp_utc": "2026-04-13T12:50:11Z", "mode": "train", "global_step": 42, "epoch": 0.004218985434455048, "loss": -0.0594, "grad_norm": 18.592884063720703, "learning_rate": 9.875757575757576e-06, "num_tokens": 77446.0, "completions/mean_length": 39.375, "completions/min_length": 29.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.375, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.7009751200675964, "rewards/meter/std": 0.37204018235206604, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9971165657043457, "rewards/repeat_soft/std": 0.003871307708323002, "rewards/judge_quality/mean": 0.32499998807907104, "rewards/judge_quality/std": 0.2537715435028076, "rewards/total_composite/mean": 0.2689000368118286, "rewards/total_composite/std": 0.2827819287776947, "reward": 0.2689000368118286, "reward_std": 0.2827818989753723, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2534762918949127, "sampling/sampling_logp_difference/max": 2.5374059677124023, "sampling/importance_sampling_ratio/min": 0.07907124608755112, "sampling/importance_sampling_ratio/mean": 1.0052545070648193, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2823383659124374, "clip_ratio/low_mean": 0.19048691913485527, "clip_ratio/low_min": 0.19048691913485527, "clip_ratio/high_mean": 0.021276595070958138, "clip_ratio/high_max": 0.021276595070958138, "clip_ratio/region_mean": 0.2117635142058134, "reward_total_mean": 0.2689000368118286, "reward_meter_mean": 0.7009751200675964, "reward_meter_std": 0.37204018235206604, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9971165657043457, "reward_repeat_soft_std": 0.003871307708323002, "reward_judge_quality_mean": 0.32499998807907104, "reward_judge_quality_std": 0.2537715435028076, "reward_total_composite_mean": 0.2689000368118286, "reward_total_composite_std": 0.2827819287776947} {"timestamp_utc": "2026-04-13T12:50:17Z", "mode": "train", "global_step": 43, "epoch": 0.004319437468608739, "loss": 0.0871, "grad_norm": 14.111536026000977, "learning_rate": 9.872727272727274e-06, "num_tokens": 79334.0, "completions/mean_length": 61.0, "completions/min_length": 28.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.6533817052841187, "rewards/meter/std": 0.4500085711479187, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9895681142807007, "rewards/repeat_soft/std": 0.012946262955665588, "rewards/judge_quality/mean": 0.4025000035762787, "rewards/judge_quality/std": 0.19099363684654236, "rewards/total_composite/mean": 0.3281722664833069, "rewards/total_composite/std": 0.24415451288223267, "reward": 0.3281722664833069, "reward_std": 0.24415451288223267, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19619974493980408, "sampling/sampling_logp_difference/max": 1.7581138610839844, "sampling/importance_sampling_ratio/min": 0.17236967384815216, "sampling/importance_sampling_ratio/mean": 1.019789457321167, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8827890306711197, "clip_ratio/low_mean": 0.1028458047658205, "clip_ratio/low_min": 0.1028458047658205, "clip_ratio/high_mean": 0.08642019052058458, "clip_ratio/high_max": 0.08642019052058458, "clip_ratio/region_mean": 0.1892659952864051, "reward_total_mean": 0.3281722664833069, "reward_meter_mean": 0.6533817052841187, "reward_meter_std": 0.4500085711479187, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9895681142807007, "reward_repeat_soft_std": 0.012946262955665588, "reward_judge_quality_mean": 0.4025000035762787, "reward_judge_quality_std": 0.19099363684654236, "reward_total_composite_mean": 0.3281722664833069, "reward_total_composite_std": 0.24415451288223267} {"timestamp_utc": "2026-04-13T12:50:24Z", "mode": "train", "global_step": 44, "epoch": 0.004419889502762431, "loss": 0.0959, "grad_norm": 11.349555015563965, "learning_rate": 9.869696969696971e-06, "num_tokens": 81552.0, "completions/mean_length": 97.25, "completions/min_length": 65.0, "completions/max_length": 142.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 97.25, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 142.0, "rewards/meter/mean": 0.34926965832710266, "rewards/meter/std": 0.3050661087036133, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.18898223340511322, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9984719753265381, "rewards/repeat_soft/std": 0.001414117286913097, "rewards/judge_quality/mean": 0.2487499862909317, "rewards/judge_quality/std": 0.0699872374534607, "rewards/total_composite/mean": 0.14352548122406006, "rewards/total_composite/std": 0.08693286031484604, "reward": 0.14352548122406006, "reward_std": 0.08693286031484604, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2462064027786255, "sampling/sampling_logp_difference/max": 1.6767797470092773, "sampling/importance_sampling_ratio/min": 0.18697510659694672, "sampling/importance_sampling_ratio/mean": 1.0362701416015625, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6043899059295654, "clip_ratio/low_mean": 0.15450546145439148, "clip_ratio/low_min": 0.15450546145439148, "clip_ratio/high_mean": 0.05760752968490124, "clip_ratio/high_max": 0.05760752968490124, "clip_ratio/region_mean": 0.21211299113929272, "reward_total_mean": 0.14352548122406006, "reward_meter_mean": 0.34926965832710266, "reward_meter_std": 0.3050661087036133, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.18898223340511322, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9984719753265381, "reward_repeat_soft_std": 0.001414117286913097, "reward_judge_quality_mean": 0.2487499862909317, "reward_judge_quality_std": 0.0699872374534607, "reward_total_composite_mean": 0.14352548122406006, "reward_total_composite_std": 0.08693286031484604} {"timestamp_utc": "2026-04-13T12:50:30Z", "mode": "train", "global_step": 45, "epoch": 0.0045203415369161224, "loss": 0.0246, "grad_norm": 19.27265167236328, "learning_rate": 9.866666666666668e-06, "num_tokens": 83060.0, "completions/mean_length": 37.5, "completions/min_length": 31.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.5, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.8940512537956238, "rewards/meter/std": 0.2674654424190521, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9911835193634033, "rewards/repeat_soft/std": 0.010344765149056911, "rewards/judge_quality/mean": 0.3737500011920929, "rewards/judge_quality/std": 0.13721074163913727, "rewards/total_composite/mean": 0.3571849465370178, "rewards/total_composite/std": 0.1577400416135788, "reward": 0.3571849465370178, "reward_std": 0.1577400267124176, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2245570421218872, "sampling/sampling_logp_difference/max": 1.398192286491394, "sampling/importance_sampling_ratio/min": 0.24704314768314362, "sampling/importance_sampling_ratio/mean": 1.0358315706253052, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0312478840351105, "clip_ratio/low_mean": 0.09685325808823109, "clip_ratio/low_min": 0.09685325808823109, "clip_ratio/high_mean": 0.1374898562207818, "clip_ratio/high_max": 0.1374898562207818, "clip_ratio/region_mean": 0.2343431143090129, "reward_total_mean": 0.3571849465370178, "reward_meter_mean": 0.8940512537956238, "reward_meter_std": 0.2674654424190521, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9911835193634033, "reward_repeat_soft_std": 0.010344765149056911, "reward_judge_quality_mean": 0.3737500011920929, "reward_judge_quality_std": 0.13721074163913727, "reward_total_composite_mean": 0.3571849465370178, "reward_total_composite_std": 0.1577400416135788} {"timestamp_utc": "2026-04-13T12:50:36Z", "mode": "train", "global_step": 46, "epoch": 0.0046207935710698145, "loss": 0.0744, "grad_norm": 24.522783279418945, "learning_rate": 9.863636363636364e-06, "num_tokens": 84627.0, "completions/mean_length": 39.875, "completions/min_length": 35.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.875, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.5732947587966919, "rewards/meter/std": 0.4071579575538635, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9985332489013672, "rewards/repeat_soft/std": 0.002732667839154601, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.1403057873249054, "rewards/total_composite/mean": 0.22148951888084412, "rewards/total_composite/std": 0.06108399108052254, "reward": 0.22148951888084412, "reward_std": 0.06108398735523224, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2159995138645172, "sampling/sampling_logp_difference/max": 1.9328289031982422, "sampling/importance_sampling_ratio/min": 0.14473818242549896, "sampling/importance_sampling_ratio/mean": 1.0379143953323364, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3000453412532806, "clip_ratio/low_mean": 0.0911612194031477, "clip_ratio/low_min": 0.0911612194031477, "clip_ratio/high_mean": 0.10591182392090559, "clip_ratio/high_max": 0.10591182392090559, "clip_ratio/region_mean": 0.1970730433240533, "reward_total_mean": 0.22148951888084412, "reward_meter_mean": 0.5732947587966919, "reward_meter_std": 0.4071579575538635, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9985332489013672, "reward_repeat_soft_std": 0.002732667839154601, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.1403057873249054, "reward_total_composite_mean": 0.22148951888084412, "reward_total_composite_std": 0.06108399108052254} {"timestamp_utc": "2026-04-13T12:50:42Z", "mode": "train", "global_step": 47, "epoch": 0.004721245605223506, "loss": 0.1602, "grad_norm": 19.036218643188477, "learning_rate": 9.860606060606061e-06, "num_tokens": 86187.0, "completions/mean_length": 46.0, "completions/min_length": 33.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.0, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.5120313167572021, "rewards/meter/std": 0.4290798306465149, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9942026138305664, "rewards/repeat_soft/std": 0.008400977589190006, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.13845448195934296, "rewards/total_composite/mean": 0.22351449728012085, "rewards/total_composite/std": 0.0865553617477417, "reward": 0.22351449728012085, "reward_std": 0.0865553542971611, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24033474922180176, "sampling/sampling_logp_difference/max": 1.5775656700134277, "sampling/importance_sampling_ratio/min": 0.2064771205186844, "sampling/importance_sampling_ratio/mean": 1.0289047956466675, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0344381779432297, "clip_ratio/low_mean": 0.10997415706515312, "clip_ratio/low_min": 0.10997415706515312, "clip_ratio/high_mean": 0.11138846911489964, "clip_ratio/high_max": 0.11138846911489964, "clip_ratio/region_mean": 0.22136262618005276, "reward_total_mean": 0.22351449728012085, "reward_meter_mean": 0.5120313167572021, "reward_meter_std": 0.4290798306465149, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9942026138305664, "reward_repeat_soft_std": 0.008400977589190006, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.13845448195934296, "reward_total_composite_mean": 0.22351449728012085, "reward_total_composite_std": 0.0865553617477417} {"timestamp_utc": "2026-04-13T12:50:48Z", "mode": "train", "global_step": 48, "epoch": 0.004821697639377198, "loss": 0.088, "grad_norm": 29.396474838256836, "learning_rate": 9.857575757575758e-06, "num_tokens": 87821.0, "completions/mean_length": 35.25, "completions/min_length": 27.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.25, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.609480619430542, "rewards/meter/std": 0.26531553268432617, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9907858967781067, "rewards/repeat_soft/std": 0.00978943146765232, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.23217912018299103, "rewards/total_composite/mean": 0.3059554100036621, "rewards/total_composite/std": 0.1653279811143875, "reward": 0.3059554100036621, "reward_std": 0.1653279811143875, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24612577259540558, "sampling/sampling_logp_difference/max": 2.0493764877319336, "sampling/importance_sampling_ratio/min": 0.1288152039051056, "sampling/importance_sampling_ratio/mean": 0.9925991296768188, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6261170208454132, "clip_ratio/low_mean": 0.1605833861976862, "clip_ratio/low_min": 0.1605833861976862, "clip_ratio/high_mean": 0.05046432092785835, "clip_ratio/high_max": 0.05046432092785835, "clip_ratio/region_mean": 0.21104770712554455, "reward_total_mean": 0.3059554100036621, "reward_meter_mean": 0.609480619430542, "reward_meter_std": 0.26531553268432617, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9907858967781067, "reward_repeat_soft_std": 0.00978943146765232, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.23217912018299103, "reward_total_composite_mean": 0.3059554100036621, "reward_total_composite_std": 0.1653279811143875} {"timestamp_utc": "2026-04-13T12:50:56Z", "mode": "train", "global_step": 49, "epoch": 0.004922149673530889, "loss": 0.1115, "grad_norm": 14.390752792358398, "learning_rate": 9.854545454545456e-06, "num_tokens": 90330.0, "completions/mean_length": 122.625, "completions/min_length": 92.0, "completions/max_length": 157.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.625, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 157.0, "rewards/meter/mean": 0.3861793279647827, "rewards/meter/std": 0.30217719078063965, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.08625820279121399, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9951063394546509, "rewards/repeat_soft/std": 0.0040963743813335896, "rewards/judge_quality/mean": 0.28999999165534973, "rewards/judge_quality/std": 0.10849621146917343, "rewards/total_composite/mean": 0.14654898643493652, "rewards/total_composite/std": 0.08626699447631836, "reward": 0.14654898643493652, "reward_std": 0.08626698702573776, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2900907099246979, "sampling/sampling_logp_difference/max": 3.003718614578247, "sampling/importance_sampling_ratio/min": 0.04960227385163307, "sampling/importance_sampling_ratio/mean": 0.9769321084022522, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6626681834459305, "clip_ratio/low_mean": 0.10772638954222202, "clip_ratio/low_min": 0.10772638954222202, "clip_ratio/high_mean": 0.12409811466932297, "clip_ratio/high_max": 0.12409811466932297, "clip_ratio/region_mean": 0.231824504211545, "reward_total_mean": 0.14654898643493652, "reward_meter_mean": 0.3861793279647827, "reward_meter_std": 0.30217719078063965, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.08625820279121399, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9951063394546509, "reward_repeat_soft_std": 0.0040963743813335896, "reward_judge_quality_mean": 0.28999999165534973, "reward_judge_quality_std": 0.10849621146917343, "reward_total_composite_mean": 0.14654898643493652, "reward_total_composite_std": 0.08626699447631836} {"timestamp_utc": "2026-04-13T12:51:03Z", "mode": "train", "global_step": 50, "epoch": 0.005022601707684581, "loss": 0.2079, "grad_norm": 12.737960815429688, "learning_rate": 9.851515151515151e-06, "num_tokens": 92296.0, "completions/mean_length": 83.75, "completions/min_length": 57.0, "completions/max_length": 118.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 83.75, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 118.0, "rewards/meter/mean": 0.276641845703125, "rewards/meter/std": 0.22296901047229767, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9936968088150024, "rewards/repeat_soft/std": 0.005944508593529463, "rewards/judge_quality/mean": 0.25999999046325684, "rewards/judge_quality/std": 0.07010196894407272, "rewards/total_composite/mean": 0.14282579720020294, "rewards/total_composite/std": 0.07114465534687042, "reward": 0.14282579720020294, "reward_std": 0.07114465534687042, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22888238728046417, "sampling/sampling_logp_difference/max": 1.416872501373291, "sampling/importance_sampling_ratio/min": 0.24247115850448608, "sampling/importance_sampling_ratio/mean": 1.0475715398788452, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.287784308195114, "clip_ratio/low_mean": 0.13792912662029266, "clip_ratio/low_min": 0.13792912662029266, "clip_ratio/high_mean": 0.077578354626894, "clip_ratio/high_max": 0.077578354626894, "clip_ratio/region_mean": 0.21550748124718666, "reward_total_mean": 0.14282579720020294, "reward_meter_mean": 0.276641845703125, "reward_meter_std": 0.22296901047229767, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9936968088150024, "reward_repeat_soft_std": 0.005944508593529463, "reward_judge_quality_mean": 0.25999999046325684, "reward_judge_quality_std": 0.07010196894407272, "reward_total_composite_mean": 0.14282579720020294, "reward_total_composite_std": 0.07114465534687042} {"timestamp_utc": "2026-04-13T12:52:02Z", "mode": "eval", "global_step": 50, "epoch": 0.005022601707684581, "eval_loss": NaN, "eval_runtime": 58.292, "eval_samples_per_second": 1.372, "eval_steps_per_second": 0.172, "eval_num_tokens": 92296.0, "eval_completions/mean_length": 78.7375, "eval_completions/min_length": 29.7, "eval_completions/max_length": 145.5, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 78.7375, "eval_completions/min_terminated_length": 29.7, "eval_completions/max_terminated_length": 145.5, "eval_rewards/meter/mean": 0.5631425261497498, "eval_rewards/meter/std": 0.3836209237575531, "eval_rewards/count_adherence/mean": 0.9606249868869782, "eval_rewards/count_adherence/std": 0.07844334654510021, "eval_rewards/hard_gate/mean": 0.9125, "eval_rewards/hard_gate/std": 0.2230676978826523, "eval_rewards/repeat_soft/mean": 0.9909339249134064, "eval_rewards/repeat_soft/std": 0.015210337797179818, "eval_rewards/judge_quality/mean": 0.2937499925494194, "eval_rewards/judge_quality/std": 0.13265312099829316, "eval_rewards/total_composite/mean": 0.2063873901963234, "eval_rewards/total_composite/std": 0.15400917865335942, "eval_reward": 0.2063873901963234, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.14399238526821137, "eval_sampling/sampling_logp_difference/max": 1.1252060890197755, "eval_sampling/importance_sampling_ratio/min": 0.32608655989170077, "eval_sampling/importance_sampling_ratio/mean": 1.0359672784805298, "eval_sampling/importance_sampling_ratio/max": 1.5399682164192199, "eval_entropy": 2.1157538652420045, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.2063873901963234, "eval_reward_meter_mean": 0.5631425261497498, "eval_reward_meter_std": 0.3836209237575531, "eval_reward_count_adherence_mean": 0.9606249868869782, "eval_reward_count_adherence_std": 0.07844334654510021, "eval_reward_hard_gate_mean": 0.9125, "eval_reward_hard_gate_std": 0.2230676978826523, "eval_reward_repeat_soft_mean": 0.9909339249134064, "eval_reward_repeat_soft_std": 0.015210337797179818, "eval_reward_judge_quality_mean": 0.2937499925494194, "eval_reward_judge_quality_std": 0.13265312099829316, "eval_reward_total_composite_mean": 0.2063873901963234, "eval_reward_total_composite_std": 0.15400917865335942} {"timestamp_utc": "2026-04-13T12:52:11Z", "mode": "train", "global_step": 51, "epoch": 0.005123053741838272, "loss": 0.1536, "grad_norm": 13.13121223449707, "learning_rate": 9.84848484848485e-06, "num_tokens": 94078.0, "completions/mean_length": 55.75, "completions/min_length": 35.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.75, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.573480486869812, "rewards/meter/std": 0.29802122712135315, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9975300431251526, "rewards/repeat_soft/std": 0.0033512876834720373, "rewards/judge_quality/mean": 0.32374998927116394, "rewards/judge_quality/std": 0.1487027108669281, "rewards/total_composite/mean": 0.23654907941818237, "rewards/total_composite/std": 0.13750989735126495, "reward": 0.23654907941818237, "reward_std": 0.13750988245010376, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19188879430294037, "sampling/sampling_logp_difference/max": 1.367401123046875, "sampling/importance_sampling_ratio/min": 0.2547681927680969, "sampling/importance_sampling_ratio/mean": 1.0396616458892822, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1432204842567444, "clip_ratio/low_mean": 0.08935721637681127, "clip_ratio/low_min": 0.08935721637681127, "clip_ratio/high_mean": 0.04642857238650322, "clip_ratio/high_max": 0.04642857238650322, "clip_ratio/region_mean": 0.13578578876331449, "reward_total_mean": 0.23654907941818237, "reward_meter_mean": 0.573480486869812, "reward_meter_std": 0.29802122712135315, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9975300431251526, "reward_repeat_soft_std": 0.0033512876834720373, "reward_judge_quality_mean": 0.32374998927116394, "reward_judge_quality_std": 0.1487027108669281, "reward_total_composite_mean": 0.23654907941818237, "reward_total_composite_std": 0.13750989735126495} {"timestamp_utc": "2026-04-13T12:52:18Z", "mode": "train", "global_step": 52, "epoch": 0.005223505775991964, "loss": 0.1116, "grad_norm": 12.535508155822754, "learning_rate": 9.845454545454546e-06, "num_tokens": 96150.0, "completions/mean_length": 85.0, "completions/min_length": 70.0, "completions/max_length": 124.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.0, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 124.0, "rewards/meter/mean": 0.784797191619873, "rewards/meter/std": 0.23827414214611053, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.993104100227356, "rewards/repeat_soft/std": 0.004849093966186047, "rewards/judge_quality/mean": 0.32374998927116394, "rewards/judge_quality/std": 0.1487027257680893, "rewards/total_composite/mean": 0.28121325373649597, "rewards/total_composite/std": 0.15429997444152832, "reward": 0.28121325373649597, "reward_std": 0.15429995954036713, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24347689747810364, "sampling/sampling_logp_difference/max": 1.919011116027832, "sampling/importance_sampling_ratio/min": 0.1467519998550415, "sampling/importance_sampling_ratio/mean": 1.0357811450958252, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.725496292114258, "clip_ratio/low_mean": 0.14716156385838985, "clip_ratio/low_min": 0.14716156385838985, "clip_ratio/high_mean": 0.07188026420772076, "clip_ratio/high_max": 0.07188026420772076, "clip_ratio/region_mean": 0.2190418280661106, "reward_total_mean": 0.28121325373649597, "reward_meter_mean": 0.784797191619873, "reward_meter_std": 0.23827414214611053, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.993104100227356, "reward_repeat_soft_std": 0.004849093966186047, "reward_judge_quality_mean": 0.32374998927116394, "reward_judge_quality_std": 0.1487027257680893, "reward_total_composite_mean": 0.28121325373649597, "reward_total_composite_std": 0.15429997444152832} {"timestamp_utc": "2026-04-13T12:52:25Z", "mode": "train", "global_step": 53, "epoch": 0.005323957810145655, "loss": 0.0093, "grad_norm": 22.459810256958008, "learning_rate": 9.842424242424243e-06, "num_tokens": 97682.0, "completions/mean_length": 38.5, "completions/min_length": 26.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.5, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.6416175365447998, "rewards/meter/std": 0.4357673227787018, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9862592816352844, "rewards/repeat_soft/std": 0.022048698738217354, "rewards/judge_quality/mean": 0.30250000953674316, "rewards/judge_quality/std": 0.09808888286352158, "rewards/total_composite/mean": 0.23128661513328552, "rewards/total_composite/std": 0.11701280623674393, "reward": 0.23128661513328552, "reward_std": 0.11701279133558273, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2023971527814865, "sampling/sampling_logp_difference/max": 2.5468854904174805, "sampling/importance_sampling_ratio/min": 0.07832523435354233, "sampling/importance_sampling_ratio/mean": 0.9859792590141296, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5731562711298466, "clip_ratio/low_mean": 0.10033906158059835, "clip_ratio/low_min": 0.10033906158059835, "clip_ratio/high_mean": 0.04652255633845925, "clip_ratio/high_max": 0.04652255633845925, "clip_ratio/region_mean": 0.1468616179190576, "reward_total_mean": 0.23128661513328552, "reward_meter_mean": 0.6416175365447998, "reward_meter_std": 0.4357673227787018, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9862592816352844, "reward_repeat_soft_std": 0.022048698738217354, "reward_judge_quality_mean": 0.30250000953674316, "reward_judge_quality_std": 0.09808888286352158, "reward_total_composite_mean": 0.23128661513328552, "reward_total_composite_std": 0.11701280623674393} {"timestamp_utc": "2026-04-13T12:52:33Z", "mode": "train", "global_step": 54, "epoch": 0.005424409844299347, "loss": 0.1895, "grad_norm": 13.117853164672852, "learning_rate": 9.83939393939394e-06, "num_tokens": 99998.0, "completions/mean_length": 112.5, "completions/min_length": 77.0, "completions/max_length": 185.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.5, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 185.0, "rewards/meter/mean": 0.29589852690696716, "rewards/meter/std": 0.34566956758499146, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9958626627922058, "rewards/repeat_soft/std": 0.00357510126195848, "rewards/judge_quality/mean": 0.26874998211860657, "rewards/judge_quality/std": 0.09523466974496841, "rewards/total_composite/mean": 0.14415407180786133, "rewards/total_composite/std": 0.06718795001506805, "reward": 0.14415407180786133, "reward_std": 0.06718795001506805, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22546643018722534, "sampling/sampling_logp_difference/max": 1.6960043907165527, "sampling/importance_sampling_ratio/min": 0.18341492116451263, "sampling/importance_sampling_ratio/mean": 1.0321494340896606, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1551734805107117, "clip_ratio/low_mean": 0.10768418572843075, "clip_ratio/low_min": 0.10768418572843075, "clip_ratio/high_mean": 0.1142218578606844, "clip_ratio/high_max": 0.1142218578606844, "clip_ratio/region_mean": 0.22190604358911514, "reward_total_mean": 0.14415407180786133, "reward_meter_mean": 0.29589852690696716, "reward_meter_std": 0.34566956758499146, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9958626627922058, "reward_repeat_soft_std": 0.00357510126195848, "reward_judge_quality_mean": 0.26874998211860657, "reward_judge_quality_std": 0.09523466974496841, "reward_total_composite_mean": 0.14415407180786133, "reward_total_composite_std": 0.06718795001506805} {"timestamp_utc": "2026-04-13T12:52:39Z", "mode": "train", "global_step": 55, "epoch": 0.0055248618784530384, "loss": 0.037, "grad_norm": 17.670133590698242, "learning_rate": 9.836363636363637e-06, "num_tokens": 101550.0, "completions/mean_length": 27.0, "completions/min_length": 19.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.0, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.4525359272956848, "rewards/meter/std": 0.3512803316116333, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.48125001788139343, "rewards/judge_quality/std": 0.2924496531486511, "rewards/total_composite/mean": 0.32188645005226135, "rewards/total_composite/std": 0.25953197479248047, "reward": 0.32188645005226135, "reward_std": 0.2595319449901581, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1912689059972763, "sampling/sampling_logp_difference/max": 1.9762558937072754, "sampling/importance_sampling_ratio/min": 0.13858714699745178, "sampling/importance_sampling_ratio/mean": 1.041345238685608, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.816194623708725, "clip_ratio/low_mean": 0.13259341288357973, "clip_ratio/low_min": 0.13259341288357973, "clip_ratio/high_mean": 0.08467167429625988, "clip_ratio/high_max": 0.08467167429625988, "clip_ratio/region_mean": 0.2172650871798396, "reward_total_mean": 0.32188645005226135, "reward_meter_mean": 0.4525359272956848, "reward_meter_std": 0.3512803316116333, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.48125001788139343, "reward_judge_quality_std": 0.2924496531486511, "reward_total_composite_mean": 0.32188645005226135, "reward_total_composite_std": 0.25953197479248047} {"timestamp_utc": "2026-04-13T12:52:48Z", "mode": "train", "global_step": 56, "epoch": 0.0056253139126067305, "loss": -0.0197, "grad_norm": 19.314542770385742, "learning_rate": 9.833333333333333e-06, "num_tokens": 103028.0, "completions/mean_length": 26.75, "completions/min_length": 19.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.75, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.3456041216850281, "rewards/meter/std": 0.446493923664093, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.25, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.14225441217422485, "rewards/total_composite/std": 0.07255525887012482, "reward": 0.14225441217422485, "reward_std": 0.07255525887012482, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22346875071525574, "sampling/sampling_logp_difference/max": 1.535238265991211, "sampling/importance_sampling_ratio/min": 0.21540437638759613, "sampling/importance_sampling_ratio/mean": 0.9892282485961914, "sampling/importance_sampling_ratio/max": 1.902657151222229, "entropy": 1.9158771336078644, "clip_ratio/low_mean": 0.13819932006299496, "clip_ratio/low_min": 0.13819932006299496, "clip_ratio/high_mean": 0.11600378900766373, "clip_ratio/high_max": 0.11600378900766373, "clip_ratio/region_mean": 0.2542031090706587, "reward_total_mean": 0.14225441217422485, "reward_meter_mean": 0.3456041216850281, "reward_meter_std": 0.446493923664093, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.25, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.14225441217422485, "reward_total_composite_std": 0.07255525887012482} {"timestamp_utc": "2026-04-13T12:52:55Z", "mode": "train", "global_step": 57, "epoch": 0.005725765946760422, "loss": 0.1387, "grad_norm": 10.624090194702148, "learning_rate": 9.830303030303032e-06, "num_tokens": 105143.0, "completions/mean_length": 94.375, "completions/min_length": 56.0, "completions/max_length": 131.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 94.375, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.7765734195709229, "rewards/meter/std": 0.2771216928958893, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9940077662467957, "rewards/repeat_soft/std": 0.009384247474372387, "rewards/judge_quality/mean": 0.28125, "rewards/judge_quality/std": 0.0882265716791153, "rewards/total_composite/mean": 0.2469685971736908, "rewards/total_composite/std": 0.10987038910388947, "reward": 0.2469685971736908, "reward_std": 0.10987038165330887, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2192666232585907, "sampling/sampling_logp_difference/max": 1.433286190032959, "sampling/importance_sampling_ratio/min": 0.25009241700172424, "sampling/importance_sampling_ratio/mean": 1.0092370510101318, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.317024514079094, "clip_ratio/low_mean": 0.11426695715636015, "clip_ratio/low_min": 0.11426695715636015, "clip_ratio/high_mean": 0.09074284695088863, "clip_ratio/high_max": 0.09074284695088863, "clip_ratio/region_mean": 0.20500980410724878, "reward_total_mean": 0.2469685971736908, "reward_meter_mean": 0.7765734195709229, "reward_meter_std": 0.2771216928958893, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9940077662467957, "reward_repeat_soft_std": 0.009384247474372387, "reward_judge_quality_mean": 0.28125, "reward_judge_quality_std": 0.0882265716791153, "reward_total_composite_mean": 0.2469685971736908, "reward_total_composite_std": 0.10987038910388947} {"timestamp_utc": "2026-04-13T12:53:02Z", "mode": "train", "global_step": 58, "epoch": 0.005826217980914114, "loss": 0.0149, "grad_norm": 10.558327674865723, "learning_rate": 9.827272727272729e-06, "num_tokens": 107309.0, "completions/mean_length": 87.75, "completions/min_length": 63.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.75, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.401488721370697, "rewards/meter/std": 0.3869341313838959, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9952694177627563, "rewards/repeat_soft/std": 0.00658520869910717, "rewards/judge_quality/mean": 0.22749999165534973, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.13977158069610596, "rewards/total_composite/std": 0.061555054038763046, "reward": 0.13977158069610596, "reward_std": 0.061555054038763046, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23097935318946838, "sampling/sampling_logp_difference/max": 1.7834727764129639, "sampling/importance_sampling_ratio/min": 0.1680535227060318, "sampling/importance_sampling_ratio/mean": 1.0329930782318115, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.427265539765358, "clip_ratio/low_mean": 0.1279610563069582, "clip_ratio/low_min": 0.1279610563069582, "clip_ratio/high_mean": 0.09140318259596825, "clip_ratio/high_max": 0.09140318259596825, "clip_ratio/region_mean": 0.21936423890292645, "reward_total_mean": 0.13977158069610596, "reward_meter_mean": 0.401488721370697, "reward_meter_std": 0.3869341313838959, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9952694177627563, "reward_repeat_soft_std": 0.00658520869910717, "reward_judge_quality_mean": 0.22749999165534973, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.13977158069610596, "reward_total_composite_std": 0.061555054038763046} {"timestamp_utc": "2026-04-13T12:53:08Z", "mode": "train", "global_step": 59, "epoch": 0.005926670015067805, "loss": -0.0049, "grad_norm": 15.45472526550293, "learning_rate": 9.824242424242425e-06, "num_tokens": 109189.0, "completions/mean_length": 56.0, "completions/min_length": 34.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.0, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.6033459901809692, "rewards/meter/std": 0.41086646914482117, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9966139793395996, "rewards/repeat_soft/std": 0.006910442374646664, "rewards/judge_quality/mean": 0.30250000953674316, "rewards/judge_quality/std": 0.20954032242298126, "rewards/total_composite/mean": 0.24199925363063812, "rewards/total_composite/std": 0.24158596992492676, "reward": 0.24199925363063812, "reward_std": 0.24158596992492676, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.224238783121109, "sampling/sampling_logp_difference/max": 1.9477157592773438, "sampling/importance_sampling_ratio/min": 0.1425994336605072, "sampling/importance_sampling_ratio/mean": 1.0293304920196533, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8733589500188828, "clip_ratio/low_mean": 0.15564925968647003, "clip_ratio/low_min": 0.15564925968647003, "clip_ratio/high_mean": 0.05659562349319458, "clip_ratio/high_max": 0.05659562349319458, "clip_ratio/region_mean": 0.2122448831796646, "reward_total_mean": 0.24199925363063812, "reward_meter_mean": 0.6033459901809692, "reward_meter_std": 0.41086646914482117, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9966139793395996, "reward_repeat_soft_std": 0.006910442374646664, "reward_judge_quality_mean": 0.30250000953674316, "reward_judge_quality_std": 0.20954032242298126, "reward_total_composite_mean": 0.24199925363063812, "reward_total_composite_std": 0.24158596992492676} {"timestamp_utc": "2026-04-13T12:53:15Z", "mode": "train", "global_step": 60, "epoch": 0.006027122049221497, "loss": 0.0373, "grad_norm": 17.19390869140625, "learning_rate": 9.821212121212122e-06, "num_tokens": 110693.0, "completions/mean_length": 43.0, "completions/min_length": 33.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.0, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.8154430389404297, "rewards/meter/std": 0.24470537900924683, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9848794937133789, "rewards/repeat_soft/std": 0.03346961736679077, "rewards/judge_quality/mean": 0.36000001430511475, "rewards/judge_quality/std": 0.17188036441802979, "rewards/total_composite/mean": 0.30636218190193176, "rewards/total_composite/std": 0.13998998701572418, "reward": 0.30636218190193176, "reward_std": 0.13999000191688538, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20322245359420776, "sampling/sampling_logp_difference/max": 1.3733539581298828, "sampling/importance_sampling_ratio/min": 0.2532561123371124, "sampling/importance_sampling_ratio/mean": 1.0346107482910156, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8409823924303055, "clip_ratio/low_mean": 0.08293335977941751, "clip_ratio/low_min": 0.08293335977941751, "clip_ratio/high_mean": 0.07878378219902515, "clip_ratio/high_max": 0.07878378219902515, "clip_ratio/region_mean": 0.16171714197844267, "reward_total_mean": 0.30636218190193176, "reward_meter_mean": 0.8154430389404297, "reward_meter_std": 0.24470537900924683, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9848794937133789, "reward_repeat_soft_std": 0.03346961736679077, "reward_judge_quality_mean": 0.36000001430511475, "reward_judge_quality_std": 0.17188036441802979, "reward_total_composite_mean": 0.30636218190193176, "reward_total_composite_std": 0.13998998701572418} {"timestamp_utc": "2026-04-13T12:53:21Z", "mode": "train", "global_step": 61, "epoch": 0.006127574083375188, "loss": 0.0539, "grad_norm": 14.639938354492188, "learning_rate": 9.81818181818182e-06, "num_tokens": 112447.0, "completions/mean_length": 45.25, "completions/min_length": 31.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.25, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.42510533332824707, "rewards/meter/std": 0.3660593330860138, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9985795021057129, "rewards/repeat_soft/std": 0.0020165368914604187, "rewards/judge_quality/mean": 0.3812499940395355, "rewards/judge_quality/std": 0.1667708158493042, "rewards/total_composite/mean": 0.22165712714195251, "rewards/total_composite/std": 0.09564365446567535, "reward": 0.22165712714195251, "reward_std": 0.09564363956451416, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22174100577831268, "sampling/sampling_logp_difference/max": 1.689234733581543, "sampling/importance_sampling_ratio/min": 0.18466079235076904, "sampling/importance_sampling_ratio/mean": 1.0645029544830322, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2769857197999954, "clip_ratio/low_mean": 0.12808788567781448, "clip_ratio/low_min": 0.12808788567781448, "clip_ratio/high_mean": 0.11083407700061798, "clip_ratio/high_max": 0.11083407700061798, "clip_ratio/region_mean": 0.23892196267843246, "reward_total_mean": 0.22165712714195251, "reward_meter_mean": 0.42510533332824707, "reward_meter_std": 0.3660593330860138, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9985795021057129, "reward_repeat_soft_std": 0.0020165368914604187, "reward_judge_quality_mean": 0.3812499940395355, "reward_judge_quality_std": 0.1667708158493042, "reward_total_composite_mean": 0.22165712714195251, "reward_total_composite_std": 0.09564365446567535} {"timestamp_utc": "2026-04-13T12:53:29Z", "mode": "train", "global_step": 62, "epoch": 0.00622802611752888, "loss": 0.0496, "grad_norm": 10.596074104309082, "learning_rate": 9.815151515151516e-06, "num_tokens": 115297.0, "completions/mean_length": 152.25, "completions/min_length": 81.0, "completions/max_length": 193.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 152.25, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 193.0, "rewards/meter/mean": 0.3930585980415344, "rewards/meter/std": 0.19574135541915894, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9903420209884644, "rewards/repeat_soft/std": 0.024503352120518684, "rewards/judge_quality/mean": 0.2774999737739563, "rewards/judge_quality/std": 0.09035643190145493, "rewards/total_composite/mean": 0.16664311289787292, "rewards/total_composite/std": 0.06804937869310379, "reward": 0.16664311289787292, "reward_std": 0.06804937869310379, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22418749332427979, "sampling/sampling_logp_difference/max": 1.676156997680664, "sampling/importance_sampling_ratio/min": 0.18709158897399902, "sampling/importance_sampling_ratio/mean": 1.0282597541809082, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.824618175625801, "clip_ratio/low_mean": 0.1227662805467844, "clip_ratio/low_min": 0.1227662805467844, "clip_ratio/high_mean": 0.07461014576256275, "clip_ratio/high_max": 0.07461014576256275, "clip_ratio/region_mean": 0.19737642630934715, "reward_total_mean": 0.16664311289787292, "reward_meter_mean": 0.3930585980415344, "reward_meter_std": 0.19574135541915894, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9903420209884644, "reward_repeat_soft_std": 0.024503352120518684, "reward_judge_quality_mean": 0.2774999737739563, "reward_judge_quality_std": 0.09035643190145493, "reward_total_composite_mean": 0.16664311289787292, "reward_total_composite_std": 0.06804937869310379} {"timestamp_utc": "2026-04-13T12:53:36Z", "mode": "train", "global_step": 63, "epoch": 0.006328478151682571, "loss": 0.0831, "grad_norm": 39.91816711425781, "learning_rate": 9.812121212121212e-06, "num_tokens": 117048.0, "completions/mean_length": 57.875, "completions/min_length": 55.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.875, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.5123488903045654, "rewards/meter/std": 0.3575299084186554, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9971224069595337, "rewards/repeat_soft/std": 0.0021660495549440384, "rewards/judge_quality/mean": 0.3100000023841858, "rewards/judge_quality/std": 0.09304375946521759, "rewards/total_composite/mean": 0.21182453632354736, "rewards/total_composite/std": 0.08976436406373978, "reward": 0.21182453632354736, "reward_std": 0.08976437151432037, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18080253899097443, "sampling/sampling_logp_difference/max": 2.7697935104370117, "sampling/importance_sampling_ratio/min": 0.06267494708299637, "sampling/importance_sampling_ratio/mean": 0.987522542476654, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7751255873590708, "clip_ratio/low_mean": 0.0875050537288189, "clip_ratio/low_min": 0.0875050537288189, "clip_ratio/high_mean": 0.08232656121253967, "clip_ratio/high_max": 0.08232656121253967, "clip_ratio/region_mean": 0.16983161494135857, "reward_total_mean": 0.21182453632354736, "reward_meter_mean": 0.5123488903045654, "reward_meter_std": 0.3575299084186554, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9971224069595337, "reward_repeat_soft_std": 0.0021660495549440384, "reward_judge_quality_mean": 0.3100000023841858, "reward_judge_quality_std": 0.09304375946521759, "reward_total_composite_mean": 0.21182453632354736, "reward_total_composite_std": 0.08976436406373978} {"timestamp_utc": "2026-04-13T12:53:43Z", "mode": "train", "global_step": 64, "epoch": 0.006428930185836263, "loss": 0.0875, "grad_norm": 14.192243576049805, "learning_rate": 9.809090909090911e-06, "num_tokens": 119152.0, "completions/mean_length": 81.0, "completions/min_length": 71.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 81.0, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.5464773178100586, "rewards/meter/std": 0.28004148602485657, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9979227781295776, "rewards/repeat_soft/std": 0.0025919738691300154, "rewards/judge_quality/mean": 0.26999998092651367, "rewards/judge_quality/std": 0.09258200228214264, "rewards/total_composite/mean": 0.18057963252067566, "rewards/total_composite/std": 0.11420687288045883, "reward": 0.18057963252067566, "reward_std": 0.11420686542987823, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2471581995487213, "sampling/sampling_logp_difference/max": 2.2663111686706543, "sampling/importance_sampling_ratio/min": 0.10369398444890976, "sampling/importance_sampling_ratio/mean": 1.0343976020812988, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.564219117164612, "clip_ratio/low_mean": 0.10504915751516819, "clip_ratio/low_min": 0.10504915751516819, "clip_ratio/high_mean": 0.11415382288396358, "clip_ratio/high_max": 0.11415382288396358, "clip_ratio/region_mean": 0.21920298039913177, "reward_total_mean": 0.18057963252067566, "reward_meter_mean": 0.5464773178100586, "reward_meter_std": 0.28004148602485657, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9979227781295776, "reward_repeat_soft_std": 0.0025919738691300154, "reward_judge_quality_mean": 0.26999998092651367, "reward_judge_quality_std": 0.09258200228214264, "reward_total_composite_mean": 0.18057963252067566, "reward_total_composite_std": 0.11420687288045883} {"timestamp_utc": "2026-04-13T12:53:49Z", "mode": "train", "global_step": 65, "epoch": 0.0065293822199899544, "loss": 0.052, "grad_norm": 29.54388427734375, "learning_rate": 9.806060606060607e-06, "num_tokens": 120695.0, "completions/mean_length": 42.875, "completions/min_length": 30.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.875, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.5739319920539856, "rewards/meter/std": 0.36332207918167114, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9990882873535156, "rewards/repeat_soft/std": 0.002288078423589468, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.15361711382865906, "rewards/total_composite/mean": 0.2958294153213501, "rewards/total_composite/std": 0.16002905368804932, "reward": 0.2958294153213501, "reward_std": 0.16002905368804932, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2263413965702057, "sampling/sampling_logp_difference/max": 2.2635364532470703, "sampling/importance_sampling_ratio/min": 0.1039821058511734, "sampling/importance_sampling_ratio/mean": 0.9821252822875977, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9036289602518082, "clip_ratio/low_mean": 0.12230175454169512, "clip_ratio/low_min": 0.12230175454169512, "clip_ratio/high_mean": 0.046962485648691654, "clip_ratio/high_max": 0.046962485648691654, "clip_ratio/region_mean": 0.16926424019038677, "reward_total_mean": 0.2958294153213501, "reward_meter_mean": 0.5739319920539856, "reward_meter_std": 0.36332207918167114, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9990882873535156, "reward_repeat_soft_std": 0.002288078423589468, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.15361711382865906, "reward_total_composite_mean": 0.2958294153213501, "reward_total_composite_std": 0.16002905368804932} {"timestamp_utc": "2026-04-13T12:53:59Z", "mode": "train", "global_step": 66, "epoch": 0.0066298342541436465, "loss": 0.1807, "grad_norm": 12.807340621948242, "learning_rate": 9.803030303030304e-06, "num_tokens": 122245.0, "completions/mean_length": 49.75, "completions/min_length": 33.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.75, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.6284730434417725, "rewards/meter/std": 0.36383092403411865, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9980244636535645, "rewards/repeat_soft/std": 0.002252033678814769, "rewards/judge_quality/mean": 0.3024999797344208, "rewards/judge_quality/std": 0.09808888286352158, "rewards/total_composite/mean": 0.23335705697536469, "rewards/total_composite/std": 0.1112920418381691, "reward": 0.23335705697536469, "reward_std": 0.1112920418381691, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22042016685009003, "sampling/sampling_logp_difference/max": 1.378006935119629, "sampling/importance_sampling_ratio/min": 0.2520804703235626, "sampling/importance_sampling_ratio/mean": 1.0496618747711182, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.378784015774727, "clip_ratio/low_mean": 0.15572327934205532, "clip_ratio/low_min": 0.15572327934205532, "clip_ratio/high_mean": 0.048660715110599995, "clip_ratio/high_max": 0.048660715110599995, "clip_ratio/region_mean": 0.20438399445265532, "reward_total_mean": 0.23335705697536469, "reward_meter_mean": 0.6284730434417725, "reward_meter_std": 0.36383092403411865, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9980244636535645, "reward_repeat_soft_std": 0.002252033678814769, "reward_judge_quality_mean": 0.3024999797344208, "reward_judge_quality_std": 0.09808888286352158, "reward_total_composite_mean": 0.23335705697536469, "reward_total_composite_std": 0.1112920418381691} {"timestamp_utc": "2026-04-13T12:54:06Z", "mode": "train", "global_step": 67, "epoch": 0.006730286288297338, "loss": 0.198, "grad_norm": 16.643781661987305, "learning_rate": 9.800000000000001e-06, "num_tokens": 123948.0, "completions/mean_length": 53.875, "completions/min_length": 31.0, "completions/max_length": 89.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.875, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 89.0, "rewards/meter/mean": 0.8704103827476501, "rewards/meter/std": 0.14272016286849976, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9991481900215149, "rewards/repeat_soft/std": 0.0020753606222569942, "rewards/judge_quality/mean": 0.2449999898672104, "rewards/judge_quality/std": 0.0707106739282608, "rewards/total_composite/mean": 0.22029772400856018, "rewards/total_composite/std": 0.08091630041599274, "reward": 0.22029772400856018, "reward_std": 0.08091631531715393, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2308141589164734, "sampling/sampling_logp_difference/max": 1.8956928253173828, "sampling/importance_sampling_ratio/min": 0.15021422505378723, "sampling/importance_sampling_ratio/mean": 1.0245487689971924, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7529315799474716, "clip_ratio/low_mean": 0.20236334949731827, "clip_ratio/low_min": 0.20236334949731827, "clip_ratio/high_mean": 0.012096773833036423, "clip_ratio/high_max": 0.012096773833036423, "clip_ratio/region_mean": 0.2144601233303547, "reward_total_mean": 0.22029772400856018, "reward_meter_mean": 0.8704103827476501, "reward_meter_std": 0.14272016286849976, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9991481900215149, "reward_repeat_soft_std": 0.0020753606222569942, "reward_judge_quality_mean": 0.2449999898672104, "reward_judge_quality_std": 0.0707106739282608, "reward_total_composite_mean": 0.22029772400856018, "reward_total_composite_std": 0.08091630041599274} {"timestamp_utc": "2026-04-13T12:54:12Z", "mode": "train", "global_step": 68, "epoch": 0.00683073832245103, "loss": 0.0626, "grad_norm": 17.704111099243164, "learning_rate": 9.796969696969698e-06, "num_tokens": 125479.0, "completions/mean_length": 41.375, "completions/min_length": 37.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.375, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.5753023624420166, "rewards/meter/std": 0.40855905413627625, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9715557098388672, "rewards/repeat_soft/std": 0.0266946442425251, "rewards/judge_quality/mean": 0.2562499940395355, "rewards/judge_quality/std": 0.06781013309955597, "rewards/total_composite/mean": 0.17713528871536255, "rewards/total_composite/std": 0.05704084038734436, "reward": 0.17713528871536255, "reward_std": 0.05704084038734436, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1867666095495224, "sampling/sampling_logp_difference/max": 1.1181955337524414, "sampling/importance_sampling_ratio/min": 0.32686910033226013, "sampling/importance_sampling_ratio/mean": 1.0235097408294678, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.623171053826809, "clip_ratio/low_mean": 0.0751488097012043, "clip_ratio/low_min": 0.0751488097012043, "clip_ratio/high_mean": 0.08889474952593446, "clip_ratio/high_max": 0.08889474952593446, "clip_ratio/region_mean": 0.16404355922713876, "reward_total_mean": 0.17713528871536255, "reward_meter_mean": 0.5753023624420166, "reward_meter_std": 0.40855905413627625, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9715557098388672, "reward_repeat_soft_std": 0.0266946442425251, "reward_judge_quality_mean": 0.2562499940395355, "reward_judge_quality_std": 0.06781013309955597, "reward_total_composite_mean": 0.17713528871536255, "reward_total_composite_std": 0.05704084038734436} {"timestamp_utc": "2026-04-13T12:54:18Z", "mode": "train", "global_step": 69, "epoch": 0.006931190356604721, "loss": 0.1341, "grad_norm": 19.775510787963867, "learning_rate": 9.793939393939394e-06, "num_tokens": 127422.0, "completions/mean_length": 55.875, "completions/min_length": 48.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.875, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.6374300718307495, "rewards/meter/std": 0.2568950951099396, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9850781559944153, "rewards/repeat_soft/std": 0.010231668129563332, "rewards/judge_quality/mean": 0.42000001668930054, "rewards/judge_quality/std": 0.18516401946544647, "rewards/total_composite/mean": 0.3325834274291992, "rewards/total_composite/std": 0.20612144470214844, "reward": 0.3325834274291992, "reward_std": 0.20612144470214844, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24013565480709076, "sampling/sampling_logp_difference/max": 3.8115100860595703, "sampling/importance_sampling_ratio/min": 0.02211475744843483, "sampling/importance_sampling_ratio/mean": 0.9981706142425537, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4976738691329956, "clip_ratio/low_mean": 0.14025673642754555, "clip_ratio/low_min": 0.14025673642754555, "clip_ratio/high_mean": 0.0714642284438014, "clip_ratio/high_max": 0.0714642284438014, "clip_ratio/region_mean": 0.21172096487134695, "reward_total_mean": 0.3325834274291992, "reward_meter_mean": 0.6374300718307495, "reward_meter_std": 0.2568950951099396, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9850781559944153, "reward_repeat_soft_std": 0.010231668129563332, "reward_judge_quality_mean": 0.42000001668930054, "reward_judge_quality_std": 0.18516401946544647, "reward_total_composite_mean": 0.3325834274291992, "reward_total_composite_std": 0.20612144470214844} {"timestamp_utc": "2026-04-13T12:54:25Z", "mode": "train", "global_step": 70, "epoch": 0.007031642390758413, "loss": -0.0385, "grad_norm": 18.971607208251953, "learning_rate": 9.790909090909093e-06, "num_tokens": 129347.0, "completions/mean_length": 68.625, "completions/min_length": 49.0, "completions/max_length": 105.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.625, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 105.0, "rewards/meter/mean": 0.18168789148330688, "rewards/meter/std": 0.11182212084531784, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9934610724449158, "rewards/repeat_soft/std": 0.005277810152620077, "rewards/judge_quality/mean": 0.23375000059604645, "rewards/judge_quality/std": 0.025599943473935127, "rewards/total_composite/mean": 0.09653282165527344, "rewards/total_composite/std": 0.040975313633680344, "reward": 0.09653282165527344, "reward_std": 0.040975309908390045, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.27443817257881165, "sampling/sampling_logp_difference/max": 1.9071316719055176, "sampling/importance_sampling_ratio/min": 0.18012750148773193, "sampling/importance_sampling_ratio/mean": 1.0374891757965088, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.7137632966041565, "clip_ratio/low_mean": 0.05093393288552761, "clip_ratio/low_min": 0.05093393288552761, "clip_ratio/high_mean": 0.1748208962380886, "clip_ratio/high_max": 0.1748208962380886, "clip_ratio/region_mean": 0.22575482912361622, "reward_total_mean": 0.09653282165527344, "reward_meter_mean": 0.18168789148330688, "reward_meter_std": 0.11182212084531784, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9934610724449158, "reward_repeat_soft_std": 0.005277810152620077, "reward_judge_quality_mean": 0.23375000059604645, "reward_judge_quality_std": 0.025599943473935127, "reward_total_composite_mean": 0.09653282165527344, "reward_total_composite_std": 0.040975313633680344} {"timestamp_utc": "2026-04-13T12:54:32Z", "mode": "train", "global_step": 71, "epoch": 0.007132094424912104, "loss": 0.4253, "grad_norm": 27.757349014282227, "learning_rate": 9.787878787878788e-06, "num_tokens": 130972.0, "completions/mean_length": 31.125, "completions/min_length": 25.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.125, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9012611508369446, "rewards/meter/std": 0.2112041562795639, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9959396719932556, "rewards/repeat_soft/std": 0.0019785345066338778, "rewards/judge_quality/mean": 0.7450000047683716, "rewards/judge_quality/std": 0.2121320217847824, "rewards/total_composite/mean": 0.7219642400741577, "rewards/total_composite/std": 0.23895327746868134, "reward": 0.7219642400741577, "reward_std": 0.23895327746868134, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1198556199669838, "sampling/sampling_logp_difference/max": 2.6911134719848633, "sampling/importance_sampling_ratio/min": 0.06780540198087692, "sampling/importance_sampling_ratio/mean": 1.0073957443237305, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5624960474669933, "clip_ratio/low_mean": 0.03030303120613098, "clip_ratio/low_min": 0.03030303120613098, "clip_ratio/high_mean": 0.06262362655252218, "clip_ratio/high_max": 0.06262362655252218, "clip_ratio/region_mean": 0.09292665775865316, "reward_total_mean": 0.7219642400741577, "reward_meter_mean": 0.9012611508369446, "reward_meter_std": 0.2112041562795639, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9959396719932556, "reward_repeat_soft_std": 0.0019785345066338778, "reward_judge_quality_mean": 0.7450000047683716, "reward_judge_quality_std": 0.2121320217847824, "reward_total_composite_mean": 0.7219642400741577, "reward_total_composite_std": 0.23895327746868134} {"timestamp_utc": "2026-04-13T12:54:38Z", "mode": "train", "global_step": 72, "epoch": 0.007232546459065796, "loss": 0.099, "grad_norm": 15.447766304016113, "learning_rate": 9.784848484848486e-06, "num_tokens": 132900.0, "completions/mean_length": 79.0, "completions/min_length": 73.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.0, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.3447451591491699, "rewards/meter/std": 0.28595930337905884, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9931455850601196, "rewards/repeat_soft/std": 0.003910026978701353, "rewards/judge_quality/mean": 0.3100000023841858, "rewards/judge_quality/std": 0.09441548585891724, "rewards/total_composite/mean": 0.18368127942085266, "rewards/total_composite/std": 0.09182135760784149, "reward": 0.18368127942085266, "reward_std": 0.0918213501572609, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22150319814682007, "sampling/sampling_logp_difference/max": 3.541834831237793, "sampling/importance_sampling_ratio/min": 0.02896014228463173, "sampling/importance_sampling_ratio/mean": 0.9910763502120972, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.12739023193717, "clip_ratio/low_mean": 0.07343300711363554, "clip_ratio/low_min": 0.07343300711363554, "clip_ratio/high_mean": 0.08443638868629932, "clip_ratio/high_max": 0.08443638868629932, "clip_ratio/region_mean": 0.15786939579993486, "reward_total_mean": 0.18368127942085266, "reward_meter_mean": 0.3447451591491699, "reward_meter_std": 0.28595930337905884, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9931455850601196, "reward_repeat_soft_std": 0.003910026978701353, "reward_judge_quality_mean": 0.3100000023841858, "reward_judge_quality_std": 0.09441548585891724, "reward_total_composite_mean": 0.18368127942085266, "reward_total_composite_std": 0.09182135760784149} {"timestamp_utc": "2026-04-13T12:54:44Z", "mode": "train", "global_step": 73, "epoch": 0.007332998493219488, "loss": 0.1001, "grad_norm": 16.453813552856445, "learning_rate": 9.781818181818183e-06, "num_tokens": 134478.0, "completions/mean_length": 38.25, "completions/min_length": 31.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.25, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.6412177681922913, "rewards/meter/std": 0.36747169494628906, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9885964393615723, "rewards/repeat_soft/std": 0.02141972817480564, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.18715444207191467, "rewards/total_composite/mean": 0.33021342754364014, "rewards/total_composite/std": 0.21703536808490753, "reward": 0.33021342754364014, "reward_std": 0.21703538298606873, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24010418355464935, "sampling/sampling_logp_difference/max": 1.873922348022461, "sampling/importance_sampling_ratio/min": 0.15352031588554382, "sampling/importance_sampling_ratio/mean": 1.0277440547943115, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0719871819019318, "clip_ratio/low_mean": 0.07103462051600218, "clip_ratio/low_min": 0.07103462051600218, "clip_ratio/high_mean": 0.0977333365008235, "clip_ratio/high_max": 0.0977333365008235, "clip_ratio/region_mean": 0.16876795701682568, "reward_total_mean": 0.33021342754364014, "reward_meter_mean": 0.6412177681922913, "reward_meter_std": 0.36747169494628906, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9885964393615723, "reward_repeat_soft_std": 0.02141972817480564, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.18715444207191467, "reward_total_composite_mean": 0.33021342754364014, "reward_total_composite_std": 0.21703536808490753} {"timestamp_utc": "2026-04-13T12:54:51Z", "mode": "train", "global_step": 74, "epoch": 0.007433450527373179, "loss": -0.0896, "grad_norm": 11.147936820983887, "learning_rate": 9.77878787878788e-06, "num_tokens": 136049.0, "completions/mean_length": 50.375, "completions/min_length": 24.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 50.375, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.7489872574806213, "rewards/meter/std": 0.250081866979599, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9957634210586548, "rewards/repeat_soft/std": 0.005310617852956057, "rewards/judge_quality/mean": 0.6812499761581421, "rewards/judge_quality/std": 0.2307402789592743, "rewards/total_composite/mean": 0.557012677192688, "rewards/total_composite/std": 0.20911905169487, "reward": 0.557012677192688, "reward_std": 0.2091190665960312, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13287407159805298, "sampling/sampling_logp_difference/max": 1.891297698020935, "sampling/importance_sampling_ratio/min": 0.15087589621543884, "sampling/importance_sampling_ratio/mean": 1.0008474588394165, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.803953543305397, "clip_ratio/low_mean": 0.08880797028541565, "clip_ratio/low_min": 0.08880797028541565, "clip_ratio/high_mean": 0.05626888386905193, "clip_ratio/high_max": 0.05626888386905193, "clip_ratio/region_mean": 0.14507685415446758, "reward_total_mean": 0.557012677192688, "reward_meter_mean": 0.7489872574806213, "reward_meter_std": 0.250081866979599, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9957634210586548, "reward_repeat_soft_std": 0.005310617852956057, "reward_judge_quality_mean": 0.6812499761581421, "reward_judge_quality_std": 0.2307402789592743, "reward_total_composite_mean": 0.557012677192688, "reward_total_composite_std": 0.20911905169487} {"timestamp_utc": "2026-04-13T12:54:58Z", "mode": "train", "global_step": 75, "epoch": 0.007533902561526871, "loss": -0.005, "grad_norm": 11.774303436279297, "learning_rate": 9.775757575757576e-06, "num_tokens": 138121.0, "completions/mean_length": 85.0, "completions/min_length": 66.0, "completions/max_length": 115.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 115.0, "rewards/meter/mean": 0.7358171939849854, "rewards/meter/std": 0.3659661114215851, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9967468976974487, "rewards/repeat_soft/std": 0.004990824498236179, "rewards/judge_quality/mean": 0.32499998807907104, "rewards/judge_quality/std": 0.13427051901817322, "rewards/total_composite/mean": 0.26152580976486206, "rewards/total_composite/std": 0.10252867639064789, "reward": 0.26152580976486206, "reward_std": 0.10252867639064789, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23619340360164642, "sampling/sampling_logp_difference/max": 1.3284320831298828, "sampling/importance_sampling_ratio/min": 0.2648922801017761, "sampling/importance_sampling_ratio/mean": 1.0346297025680542, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.883807897567749, "clip_ratio/low_mean": 0.04873333126306534, "clip_ratio/low_min": 0.04873333126306534, "clip_ratio/high_mean": 0.14055721275508404, "clip_ratio/high_max": 0.14055721275508404, "clip_ratio/region_mean": 0.18929054401814938, "reward_total_mean": 0.26152580976486206, "reward_meter_mean": 0.7358171939849854, "reward_meter_std": 0.3659661114215851, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9967468976974487, "reward_repeat_soft_std": 0.004990824498236179, "reward_judge_quality_mean": 0.32499998807907104, "reward_judge_quality_std": 0.13427051901817322, "reward_total_composite_mean": 0.26152580976486206, "reward_total_composite_std": 0.10252867639064789} {"timestamp_utc": "2026-04-13T12:55:03Z", "mode": "train", "global_step": 76, "epoch": 0.0076343545956805625, "loss": 0.1574, "grad_norm": 25.992359161376953, "learning_rate": 9.772727272727273e-06, "num_tokens": 139528.0, "completions/mean_length": 21.875, "completions/min_length": 17.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.875, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.6581739187240601, "rewards/meter/std": 0.44663143157958984, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9404481053352356, "rewards/repeat_soft/std": 0.0446208231151104, "rewards/judge_quality/mean": 0.3424999713897705, "rewards/judge_quality/std": 0.20190167427062988, "rewards/total_composite/mean": 0.24099212884902954, "rewards/total_composite/std": 0.09951608628034592, "reward": 0.24099212884902954, "reward_std": 0.09951609373092651, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17727313935756683, "sampling/sampling_logp_difference/max": 0.8434514999389648, "sampling/importance_sampling_ratio/min": 0.4302230179309845, "sampling/importance_sampling_ratio/mean": 1.0560768842697144, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2954431772232056, "clip_ratio/low_mean": 0.051764706149697304, "clip_ratio/low_min": 0.051764706149697304, "clip_ratio/high_mean": 0.11540627013891935, "clip_ratio/high_max": 0.11540627013891935, "clip_ratio/region_mean": 0.16717097628861666, "reward_total_mean": 0.24099212884902954, "reward_meter_mean": 0.6581739187240601, "reward_meter_std": 0.44663143157958984, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9404481053352356, "reward_repeat_soft_std": 0.0446208231151104, "reward_judge_quality_mean": 0.3424999713897705, "reward_judge_quality_std": 0.20190167427062988, "reward_total_composite_mean": 0.24099212884902954, "reward_total_composite_std": 0.09951608628034592} {"timestamp_utc": "2026-04-13T12:55:10Z", "mode": "train", "global_step": 77, "epoch": 0.0077348066298342545, "loss": 0.1504, "grad_norm": 17.336942672729492, "learning_rate": 9.76969696969697e-06, "num_tokens": 141097.0, "completions/mean_length": 40.125, "completions/min_length": 29.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.125, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.5049496293067932, "rewards/meter/std": 0.3687618672847748, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9967241883277893, "rewards/repeat_soft/std": 0.004613413475453854, "rewards/judge_quality/mean": 0.38499999046325684, "rewards/judge_quality/std": 0.1629198044538498, "rewards/total_composite/mean": 0.2808726131916046, "rewards/total_composite/std": 0.2017921358346939, "reward": 0.2808726131916046, "reward_std": 0.2017921358346939, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1849796026945114, "sampling/sampling_logp_difference/max": 1.5303640365600586, "sampling/importance_sampling_ratio/min": 0.21645687520503998, "sampling/importance_sampling_ratio/mean": 0.9946142435073853, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4863542169332504, "clip_ratio/low_mean": 0.11449978314340115, "clip_ratio/low_min": 0.11449978314340115, "clip_ratio/high_mean": 0.0906324926763773, "clip_ratio/high_max": 0.0906324926763773, "clip_ratio/region_mean": 0.20513227581977844, "reward_total_mean": 0.2808726131916046, "reward_meter_mean": 0.5049496293067932, "reward_meter_std": 0.3687618672847748, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9967241883277893, "reward_repeat_soft_std": 0.004613413475453854, "reward_judge_quality_mean": 0.38499999046325684, "reward_judge_quality_std": 0.1629198044538498, "reward_total_composite_mean": 0.2808726131916046, "reward_total_composite_std": 0.2017921358346939} {"timestamp_utc": "2026-04-13T12:55:17Z", "mode": "train", "global_step": 78, "epoch": 0.007835258663987946, "loss": 0.068, "grad_norm": 11.861577987670898, "learning_rate": 9.766666666666667e-06, "num_tokens": 143395.0, "completions/mean_length": 92.25, "completions/min_length": 61.0, "completions/max_length": 131.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 92.25, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.7410093545913696, "rewards/meter/std": 0.14935754239559174, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9929521083831787, "rewards/repeat_soft/std": 0.010279295966029167, "rewards/judge_quality/mean": 0.3974999785423279, "rewards/judge_quality/std": 0.179344043135643, "rewards/total_composite/mean": 0.33023709058761597, "rewards/total_composite/std": 0.16106951236724854, "reward": 0.33023709058761597, "reward_std": 0.16106951236724854, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22217519581317902, "sampling/sampling_logp_difference/max": 2.2308349609375, "sampling/importance_sampling_ratio/min": 0.10743868350982666, "sampling/importance_sampling_ratio/mean": 1.0363174676895142, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6303199976682663, "clip_ratio/low_mean": 0.11399947199970484, "clip_ratio/low_min": 0.11399947199970484, "clip_ratio/high_mean": 0.09098470397293568, "clip_ratio/high_max": 0.09098470397293568, "clip_ratio/region_mean": 0.20498417597264051, "reward_total_mean": 0.33023709058761597, "reward_meter_mean": 0.7410093545913696, "reward_meter_std": 0.14935754239559174, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9929521083831787, "reward_repeat_soft_std": 0.010279295966029167, "reward_judge_quality_mean": 0.3974999785423279, "reward_judge_quality_std": 0.179344043135643, "reward_total_composite_mean": 0.33023709058761597, "reward_total_composite_std": 0.16106951236724854} {"timestamp_utc": "2026-04-13T12:55:23Z", "mode": "train", "global_step": 79, "epoch": 0.007935710698141637, "loss": -0.0253, "grad_norm": 13.1495943069458, "learning_rate": 9.763636363636365e-06, "num_tokens": 145468.0, "completions/mean_length": 82.125, "completions/min_length": 55.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.125, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.29229092597961426, "rewards/meter/std": 0.2564309537410736, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9963474273681641, "rewards/repeat_soft/std": 0.0026311695110052824, "rewards/judge_quality/mean": 0.22374999523162842, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.10872744768857956, "rewards/total_composite/std": 0.05904335901141167, "reward": 0.10872744768857956, "reward_std": 0.05904335528612137, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20582804083824158, "sampling/sampling_logp_difference/max": 1.7601685523986816, "sampling/importance_sampling_ratio/min": 0.17201586067676544, "sampling/importance_sampling_ratio/mean": 1.030929684638977, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0478455275297165, "clip_ratio/low_mean": 0.13159164413809776, "clip_ratio/low_min": 0.13159164413809776, "clip_ratio/high_mean": 0.09389667399227619, "clip_ratio/high_max": 0.09389667399227619, "clip_ratio/region_mean": 0.22548831813037395, "reward_total_mean": 0.10872744768857956, "reward_meter_mean": 0.29229092597961426, "reward_meter_std": 0.2564309537410736, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9963474273681641, "reward_repeat_soft_std": 0.0026311695110052824, "reward_judge_quality_mean": 0.22374999523162842, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.10872744768857956, "reward_total_composite_std": 0.05904335901141167} {"timestamp_utc": "2026-04-13T12:55:30Z", "mode": "train", "global_step": 80, "epoch": 0.00803616273229533, "loss": 0.0505, "grad_norm": 14.630961418151855, "learning_rate": 9.760606060606062e-06, "num_tokens": 147164.0, "completions/mean_length": 43.0, "completions/min_length": 29.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.0, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.8201456665992737, "rewards/meter/std": 0.3403693437576294, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9994844198226929, "rewards/repeat_soft/std": 0.0014582598814740777, "rewards/judge_quality/mean": 0.3062500059604645, "rewards/judge_quality/std": 0.14302222430706024, "rewards/total_composite/mean": 0.2373991310596466, "rewards/total_composite/std": 0.17332343757152557, "reward": 0.2373991310596466, "reward_std": 0.17332343757152557, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22331182658672333, "sampling/sampling_logp_difference/max": 1.3764514923095703, "sampling/importance_sampling_ratio/min": 0.2524728775024414, "sampling/importance_sampling_ratio/mean": 1.042371392250061, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.39580474793911, "clip_ratio/low_mean": 0.1384577490389347, "clip_ratio/low_min": 0.1384577490389347, "clip_ratio/high_mean": 0.09441963024437428, "clip_ratio/high_max": 0.09441963024437428, "clip_ratio/region_mean": 0.23287737928330898, "reward_total_mean": 0.2373991310596466, "reward_meter_mean": 0.8201456665992737, "reward_meter_std": 0.3403693437576294, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9994844198226929, "reward_repeat_soft_std": 0.0014582598814740777, "reward_judge_quality_mean": 0.3062500059604645, "reward_judge_quality_std": 0.14302222430706024, "reward_total_composite_mean": 0.2373991310596466, "reward_total_composite_std": 0.17332343757152557} {"timestamp_utc": "2026-04-13T12:55:37Z", "mode": "train", "global_step": 81, "epoch": 0.008136614766449021, "loss": 0.043, "grad_norm": 11.774923324584961, "learning_rate": 9.757575757575758e-06, "num_tokens": 149486.0, "completions/mean_length": 80.25, "completions/min_length": 72.0, "completions/max_length": 100.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.25, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 100.0, "rewards/meter/mean": 0.9092499017715454, "rewards/meter/std": 0.14935287833213806, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9916456341743469, "rewards/repeat_soft/std": 0.00773578928783536, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09227287769317627, "rewards/total_composite/mean": 0.3116473853588104, "rewards/total_composite/std": 0.08176867663860321, "reward": 0.3116473853588104, "reward_std": 0.08176867663860321, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2346639186143875, "sampling/sampling_logp_difference/max": 1.6828041076660156, "sampling/importance_sampling_ratio/min": 0.23151129484176636, "sampling/importance_sampling_ratio/mean": 1.060604214668274, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.0485707223415375, "clip_ratio/low_mean": 0.13378658518195152, "clip_ratio/low_min": 0.13378658518195152, "clip_ratio/high_mean": 0.09180899523198605, "clip_ratio/high_max": 0.09180899523198605, "clip_ratio/region_mean": 0.22559558041393757, "reward_total_mean": 0.3116473853588104, "reward_meter_mean": 0.9092499017715454, "reward_meter_std": 0.14935287833213806, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9916456341743469, "reward_repeat_soft_std": 0.00773578928783536, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09227287769317627, "reward_total_composite_mean": 0.3116473853588104, "reward_total_composite_std": 0.08176867663860321} {"timestamp_utc": "2026-04-13T12:55:43Z", "mode": "train", "global_step": 82, "epoch": 0.008237066800602712, "loss": 0.1046, "grad_norm": 18.41388511657715, "learning_rate": 9.754545454545455e-06, "num_tokens": 150996.0, "completions/mean_length": 39.75, "completions/min_length": 31.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.75, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.3590666651725769, "rewards/meter/std": 0.34565114974975586, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9989273548126221, "rewards/repeat_soft/std": 0.002356387674808502, "rewards/judge_quality/mean": 0.22374999523162842, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.13183966279029846, "rewards/total_composite/std": 0.05637011304497719, "reward": 0.13183966279029846, "reward_std": 0.05637011677026749, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2514374256134033, "sampling/sampling_logp_difference/max": 1.5833864212036133, "sampling/importance_sampling_ratio/min": 0.20527875423431396, "sampling/importance_sampling_ratio/mean": 1.02720308303833, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.521538481116295, "clip_ratio/low_mean": 0.07123085763305426, "clip_ratio/low_min": 0.07123085763305426, "clip_ratio/high_mean": 0.11304819025099277, "clip_ratio/high_max": 0.11304819025099277, "clip_ratio/region_mean": 0.18427904788404703, "reward_total_mean": 0.13183966279029846, "reward_meter_mean": 0.3590666651725769, "reward_meter_std": 0.34565114974975586, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9989273548126221, "reward_repeat_soft_std": 0.002356387674808502, "reward_judge_quality_mean": 0.22374999523162842, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.13183966279029846, "reward_total_composite_std": 0.05637011304497719} {"timestamp_utc": "2026-04-13T12:55:49Z", "mode": "train", "global_step": 83, "epoch": 0.008337518834756403, "loss": 0.1639, "grad_norm": 14.891986846923828, "learning_rate": 9.751515151515152e-06, "num_tokens": 152512.0, "completions/mean_length": 44.5, "completions/min_length": 35.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.5, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.760381817817688, "rewards/meter/std": 0.31577998399734497, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9976599216461182, "rewards/repeat_soft/std": 0.004242512863129377, "rewards/judge_quality/mean": 0.21875, "rewards/judge_quality/std": 0.018850916996598244, "rewards/total_composite/mean": 0.18250644207000732, "rewards/total_composite/std": 0.04083961620926857, "reward": 0.18250644207000732, "reward_std": 0.04083961620926857, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18133598566055298, "sampling/sampling_logp_difference/max": 1.790323257446289, "sampling/importance_sampling_ratio/min": 0.1669062077999115, "sampling/importance_sampling_ratio/mean": 1.0080831050872803, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5756563022732735, "clip_ratio/low_mean": 0.059422025457024574, "clip_ratio/low_min": 0.059422025457024574, "clip_ratio/high_mean": 0.09980646055191755, "clip_ratio/high_max": 0.09980646055191755, "clip_ratio/region_mean": 0.15922848600894213, "reward_total_mean": 0.18250644207000732, "reward_meter_mean": 0.760381817817688, "reward_meter_std": 0.31577998399734497, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9976599216461182, "reward_repeat_soft_std": 0.004242512863129377, "reward_judge_quality_mean": 0.21875, "reward_judge_quality_std": 0.018850916996598244, "reward_total_composite_mean": 0.18250644207000732, "reward_total_composite_std": 0.04083961620926857} {"timestamp_utc": "2026-04-13T12:55:55Z", "mode": "train", "global_step": 84, "epoch": 0.008437970868910096, "loss": 0.0318, "grad_norm": 18.084400177001953, "learning_rate": 9.74848484848485e-06, "num_tokens": 154254.0, "completions/mean_length": 36.75, "completions/min_length": 33.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.75, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.4844000041484833, "rewards/meter/std": 0.40570446848869324, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9953454732894897, "rewards/repeat_soft/std": 0.006105078384280205, "rewards/judge_quality/mean": 0.26999998092651367, "rewards/judge_quality/std": 0.1414213627576828, "rewards/total_composite/mean": 0.1721809208393097, "rewards/total_composite/std": 0.08328097313642502, "reward": 0.1721809208393097, "reward_std": 0.08328097313642502, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24323490262031555, "sampling/sampling_logp_difference/max": 2.1036877632141113, "sampling/importance_sampling_ratio/min": 0.12200567126274109, "sampling/importance_sampling_ratio/mean": 1.0523042678833008, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.703775256872177, "clip_ratio/low_mean": 0.12026222702115774, "clip_ratio/low_min": 0.12026222702115774, "clip_ratio/high_mean": 0.09044239670038223, "clip_ratio/high_max": 0.09044239670038223, "clip_ratio/region_mean": 0.21070462372153997, "reward_total_mean": 0.1721809208393097, "reward_meter_mean": 0.4844000041484833, "reward_meter_std": 0.40570446848869324, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9953454732894897, "reward_repeat_soft_std": 0.006105078384280205, "reward_judge_quality_mean": 0.26999998092651367, "reward_judge_quality_std": 0.1414213627576828, "reward_total_composite_mean": 0.1721809208393097, "reward_total_composite_std": 0.08328097313642502} {"timestamp_utc": "2026-04-13T12:56:01Z", "mode": "train", "global_step": 85, "epoch": 0.008538422903063787, "loss": 0.1219, "grad_norm": 30.80690574645996, "learning_rate": 9.745454545454547e-06, "num_tokens": 155560.0, "completions/mean_length": 18.25, "completions/min_length": 14.0, "completions/max_length": 24.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 18.25, "completions/min_terminated_length": 14.0, "completions/max_terminated_length": 24.0, "rewards/meter/mean": 0.7818843722343445, "rewards/meter/std": 0.324420303106308, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9512310028076172, "rewards/repeat_soft/std": 0.022036533802747726, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.29293006658554077, "rewards/total_composite/std": 0.12112832814455032, "reward": 0.29293006658554077, "reward_std": 0.12112832069396973, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2320132553577423, "sampling/sampling_logp_difference/max": 1.6422712802886963, "sampling/importance_sampling_ratio/min": 0.19353996217250824, "sampling/importance_sampling_ratio/mean": 1.038288950920105, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9580261558294296, "clip_ratio/low_mean": 0.07638351526111364, "clip_ratio/low_min": 0.07638351526111364, "clip_ratio/high_mean": 0.10431678220629692, "clip_ratio/high_max": 0.10431678220629692, "clip_ratio/region_mean": 0.18070029746741056, "reward_total_mean": 0.29293006658554077, "reward_meter_mean": 0.7818843722343445, "reward_meter_std": 0.324420303106308, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9512310028076172, "reward_repeat_soft_std": 0.022036533802747726, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.29293006658554077, "reward_total_composite_std": 0.12112832814455032} {"timestamp_utc": "2026-04-13T12:56:08Z", "mode": "train", "global_step": 86, "epoch": 0.008638874937217478, "loss": 0.0126, "grad_norm": 16.33660888671875, "learning_rate": 9.742424242424244e-06, "num_tokens": 157581.0, "completions/mean_length": 73.625, "completions/min_length": 66.0, "completions/max_length": 86.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.625, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.3446776866912842, "rewards/meter/std": 0.27230140566825867, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9970155954360962, "rewards/repeat_soft/std": 0.002982294885441661, "rewards/judge_quality/mean": 0.2150000035762787, "rewards/judge_quality/std": 0.014142133295536041, "rewards/total_composite/mean": 0.09810009598731995, "rewards/total_composite/std": 0.06941568851470947, "reward": 0.09810009598731995, "reward_std": 0.06941568106412888, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2284555733203888, "sampling/sampling_logp_difference/max": 2.1358413696289062, "sampling/importance_sampling_ratio/min": 0.11814513802528381, "sampling/importance_sampling_ratio/mean": 0.9742966294288635, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.226352721452713, "clip_ratio/low_mean": 0.08101782202720642, "clip_ratio/low_min": 0.08101782202720642, "clip_ratio/high_mean": 0.10740447510033846, "clip_ratio/high_max": 0.10740447510033846, "clip_ratio/region_mean": 0.18842229712754488, "reward_total_mean": 0.09810009598731995, "reward_meter_mean": 0.3446776866912842, "reward_meter_std": 0.27230140566825867, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9970155954360962, "reward_repeat_soft_std": 0.002982294885441661, "reward_judge_quality_mean": 0.2150000035762787, "reward_judge_quality_std": 0.014142133295536041, "reward_total_composite_mean": 0.09810009598731995, "reward_total_composite_std": 0.06941568851470947} {"timestamp_utc": "2026-04-13T12:56:13Z", "mode": "train", "global_step": 87, "epoch": 0.00873932697137117, "loss": 0.0745, "grad_norm": 29.98630142211914, "learning_rate": 9.739393939393941e-06, "num_tokens": 158871.0, "completions/mean_length": 20.25, "completions/min_length": 14.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 20.25, "completions/min_terminated_length": 14.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.5150893330574036, "rewards/meter/std": 0.4621303379535675, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9594135880470276, "rewards/repeat_soft/std": 0.005897748749703169, "rewards/judge_quality/mean": 0.2512499988079071, "rewards/judge_quality/std": 0.07623975723981857, "rewards/total_composite/mean": 0.16783645749092102, "rewards/total_composite/std": 0.12525562942028046, "reward": 0.16783645749092102, "reward_std": 0.12525562942028046, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2461036890745163, "sampling/sampling_logp_difference/max": 2.1723361015319824, "sampling/importance_sampling_ratio/min": 0.1139112040400505, "sampling/importance_sampling_ratio/mean": 1.0283335447311401, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.04075089097023, "clip_ratio/low_mean": 0.10652709566056728, "clip_ratio/low_min": 0.10652709566056728, "clip_ratio/high_mean": 0.10458851745352149, "clip_ratio/high_max": 0.10458851745352149, "clip_ratio/region_mean": 0.21111561311408877, "reward_total_mean": 0.16783645749092102, "reward_meter_mean": 0.5150893330574036, "reward_meter_std": 0.4621303379535675, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9594135880470276, "reward_repeat_soft_std": 0.005897748749703169, "reward_judge_quality_mean": 0.2512499988079071, "reward_judge_quality_std": 0.07623975723981857, "reward_total_composite_mean": 0.16783645749092102, "reward_total_composite_std": 0.12525562942028046} {"timestamp_utc": "2026-04-13T12:56:20Z", "mode": "train", "global_step": 88, "epoch": 0.008839779005524863, "loss": 0.0042, "grad_norm": 17.625898361206055, "learning_rate": 9.736363636363637e-06, "num_tokens": 160484.0, "completions/mean_length": 40.625, "completions/min_length": 34.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.625, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.25569644570350647, "rewards/meter/std": 0.21334248781204224, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9973743557929993, "rewards/repeat_soft/std": 0.004422957077622414, "rewards/judge_quality/mean": 0.22624999284744263, "rewards/judge_quality/std": 0.02386719174683094, "rewards/total_composite/mean": 0.11431076377630234, "rewards/total_composite/std": 0.022492822259664536, "reward": 0.11431076377630234, "reward_std": 0.022492822259664536, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20674750208854675, "sampling/sampling_logp_difference/max": 1.886871337890625, "sampling/importance_sampling_ratio/min": 0.1515451967716217, "sampling/importance_sampling_ratio/mean": 1.022700548171997, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7788608074188232, "clip_ratio/low_mean": 0.10440731886774302, "clip_ratio/low_min": 0.10440731886774302, "clip_ratio/high_mean": 0.07957121171057224, "clip_ratio/high_max": 0.07957121171057224, "clip_ratio/region_mean": 0.18397853057831526, "reward_total_mean": 0.11431076377630234, "reward_meter_mean": 0.25569644570350647, "reward_meter_std": 0.21334248781204224, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9973743557929993, "reward_repeat_soft_std": 0.004422957077622414, "reward_judge_quality_mean": 0.22624999284744263, "reward_judge_quality_std": 0.02386719174683094, "reward_total_composite_mean": 0.11431076377630234, "reward_total_composite_std": 0.022492822259664536} {"timestamp_utc": "2026-04-13T12:56:27Z", "mode": "train", "global_step": 89, "epoch": 0.008940231039678554, "loss": 0.0473, "grad_norm": 14.281908988952637, "learning_rate": 9.733333333333334e-06, "num_tokens": 162532.0, "completions/mean_length": 91.0, "completions/min_length": 67.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.5333120822906494, "rewards/meter/std": 0.3111574351787567, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9964537620544434, "rewards/repeat_soft/std": 0.0021019114647060633, "rewards/judge_quality/mean": 0.3487499952316284, "rewards/judge_quality/std": 0.09876921027898788, "rewards/total_composite/mean": 0.21525812149047852, "rewards/total_composite/std": 0.12606896460056305, "reward": 0.21525812149047852, "reward_std": 0.12606896460056305, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16212689876556396, "sampling/sampling_logp_difference/max": 2.8756399154663086, "sampling/importance_sampling_ratio/min": 0.056380048394203186, "sampling/importance_sampling_ratio/mean": 1.006917953491211, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7011619098484516, "clip_ratio/low_mean": 0.048041378147900105, "clip_ratio/low_min": 0.048041378147900105, "clip_ratio/high_mean": 0.08259881474077702, "clip_ratio/high_max": 0.08259881474077702, "clip_ratio/region_mean": 0.13064019288867712, "reward_total_mean": 0.21525812149047852, "reward_meter_mean": 0.5333120822906494, "reward_meter_std": 0.3111574351787567, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9964537620544434, "reward_repeat_soft_std": 0.0021019114647060633, "reward_judge_quality_mean": 0.3487499952316284, "reward_judge_quality_std": 0.09876921027898788, "reward_total_composite_mean": 0.21525812149047852, "reward_total_composite_std": 0.12606896460056305} {"timestamp_utc": "2026-04-13T12:56:33Z", "mode": "train", "global_step": 90, "epoch": 0.009040683073832245, "loss": -0.0018, "grad_norm": 29.664228439331055, "learning_rate": 9.730303030303031e-06, "num_tokens": 163974.0, "completions/mean_length": 23.25, "completions/min_length": 21.0, "completions/max_length": 27.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.25, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 27.0, "rewards/meter/mean": 0.5339425802230835, "rewards/meter/std": 0.3753529191017151, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9940358996391296, "rewards/repeat_soft/std": 0.00873249489814043, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.13845448195934296, "rewards/total_composite/mean": 0.23664940893650055, "rewards/total_composite/std": 0.14900387823581696, "reward": 0.23664940893650055, "reward_std": 0.14900384843349457, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23893314599990845, "sampling/sampling_logp_difference/max": 1.5105533599853516, "sampling/importance_sampling_ratio/min": 0.22078776359558105, "sampling/importance_sampling_ratio/mean": 0.9652258157730103, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6168277114629745, "clip_ratio/low_mean": 0.06684981845319271, "clip_ratio/low_min": 0.06684981845319271, "clip_ratio/high_mean": 0.1836034394800663, "clip_ratio/high_max": 0.1836034394800663, "clip_ratio/region_mean": 0.250453257933259, "reward_total_mean": 0.23664940893650055, "reward_meter_mean": 0.5339425802230835, "reward_meter_std": 0.3753529191017151, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9940358996391296, "reward_repeat_soft_std": 0.00873249489814043, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.13845448195934296, "reward_total_composite_mean": 0.23664940893650055, "reward_total_composite_std": 0.14900387823581696} {"timestamp_utc": "2026-04-13T12:56:39Z", "mode": "train", "global_step": 91, "epoch": 0.009141135107985936, "loss": 0.0441, "grad_norm": 13.200789451599121, "learning_rate": 9.727272727272728e-06, "num_tokens": 165909.0, "completions/mean_length": 68.875, "completions/min_length": 59.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.875, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.5148859024047852, "rewards/meter/std": 0.34434616565704346, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9969649314880371, "rewards/repeat_soft/std": 0.0031781643629074097, "rewards/judge_quality/mean": 0.3199999928474426, "rewards/judge_quality/std": 0.10690449178218842, "rewards/total_composite/mean": 0.197257399559021, "rewards/total_composite/std": 0.14142344892024994, "reward": 0.197257399559021, "reward_std": 0.14142344892024994, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24964481592178345, "sampling/sampling_logp_difference/max": 2.125476837158203, "sampling/importance_sampling_ratio/min": 0.1193760335445404, "sampling/importance_sampling_ratio/mean": 1.0264556407928467, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.708129793405533, "clip_ratio/low_mean": 0.1285439468920231, "clip_ratio/low_min": 0.1285439468920231, "clip_ratio/high_mean": 0.0763765349984169, "clip_ratio/high_max": 0.0763765349984169, "clip_ratio/region_mean": 0.20492048189044, "reward_total_mean": 0.197257399559021, "reward_meter_mean": 0.5148859024047852, "reward_meter_std": 0.34434616565704346, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9969649314880371, "reward_repeat_soft_std": 0.0031781643629074097, "reward_judge_quality_mean": 0.3199999928474426, "reward_judge_quality_std": 0.10690449178218842, "reward_total_composite_mean": 0.197257399559021, "reward_total_composite_std": 0.14142344892024994} {"timestamp_utc": "2026-04-13T12:56:46Z", "mode": "train", "global_step": 92, "epoch": 0.009241587142139629, "loss": -0.0133, "grad_norm": 19.60318374633789, "learning_rate": 9.724242424242426e-06, "num_tokens": 167577.0, "completions/mean_length": 40.5, "completions/min_length": 34.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.5, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.2892237901687622, "rewards/meter/std": 0.32355058193206787, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9792777299880981, "rewards/repeat_soft/std": 0.03071063570678234, "rewards/judge_quality/mean": 0.32625001668930054, "rewards/judge_quality/std": 0.14831794798374176, "rewards/total_composite/mean": 0.15861031413078308, "rewards/total_composite/std": 0.0482286661863327, "reward": 0.15861031413078308, "reward_std": 0.048228669911623, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23212920129299164, "sampling/sampling_logp_difference/max": 1.5110788345336914, "sampling/importance_sampling_ratio/min": 0.22067178785800934, "sampling/importance_sampling_ratio/mean": 1.0275473594665527, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.950747475028038, "clip_ratio/low_mean": 0.13828057888895273, "clip_ratio/low_min": 0.13828057888895273, "clip_ratio/high_mean": 0.10061880759894848, "clip_ratio/high_max": 0.10061880759894848, "clip_ratio/region_mean": 0.2388993864879012, "reward_total_mean": 0.15861031413078308, "reward_meter_mean": 0.2892237901687622, "reward_meter_std": 0.32355058193206787, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9792777299880981, "reward_repeat_soft_std": 0.03071063570678234, "reward_judge_quality_mean": 0.32625001668930054, "reward_judge_quality_std": 0.14831794798374176, "reward_total_composite_mean": 0.15861031413078308, "reward_total_composite_std": 0.0482286661863327} {"timestamp_utc": "2026-04-13T12:56:52Z", "mode": "train", "global_step": 93, "epoch": 0.00934203917629332, "loss": -0.1215, "grad_norm": 14.864767074584961, "learning_rate": 9.721212121212123e-06, "num_tokens": 169145.0, "completions/mean_length": 49.0, "completions/min_length": 38.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 49.0, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.7403737306594849, "rewards/meter/std": 0.40326493978500366, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9993371367454529, "rewards/repeat_soft/std": 0.0018749026348814368, "rewards/judge_quality/mean": 0.24250000715255737, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.20033590495586395, "rewards/total_composite/std": 0.061576951295137405, "reward": 0.20033590495586395, "reward_std": 0.061576951295137405, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22364959120750427, "sampling/sampling_logp_difference/max": 1.231515645980835, "sampling/importance_sampling_ratio/min": 0.29184991121292114, "sampling/importance_sampling_ratio/mean": 1.062864065170288, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.15771621465683, "clip_ratio/low_mean": 0.0690789483487606, "clip_ratio/low_min": 0.0690789483487606, "clip_ratio/high_mean": 0.15948813129216433, "clip_ratio/high_max": 0.15948813129216433, "clip_ratio/region_mean": 0.22856707964092493, "reward_total_mean": 0.20033590495586395, "reward_meter_mean": 0.7403737306594849, "reward_meter_std": 0.40326493978500366, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9993371367454529, "reward_repeat_soft_std": 0.0018749026348814368, "reward_judge_quality_mean": 0.24250000715255737, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.20033590495586395, "reward_total_composite_std": 0.061576951295137405} {"timestamp_utc": "2026-04-13T12:56:58Z", "mode": "train", "global_step": 94, "epoch": 0.009442491210447011, "loss": 0.0733, "grad_norm": 20.891071319580078, "learning_rate": 9.718181818181818e-06, "num_tokens": 170838.0, "completions/mean_length": 33.625, "completions/min_length": 31.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.625, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.40905338525772095, "rewards/meter/std": 0.38333770632743835, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9920638799667358, "rewards/repeat_soft/std": 0.010272251442074776, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.22936499118804932, "rewards/total_composite/std": 0.12850287556648254, "reward": 0.22936499118804932, "reward_std": 0.12850286066532135, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23356930911540985, "sampling/sampling_logp_difference/max": 1.6581082344055176, "sampling/importance_sampling_ratio/min": 0.19049902260303497, "sampling/importance_sampling_ratio/mean": 1.0124125480651855, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.796781674027443, "clip_ratio/low_mean": 0.13331238925457, "clip_ratio/low_min": 0.13331238925457, "clip_ratio/high_mean": 0.0598262045532465, "clip_ratio/high_max": 0.0598262045532465, "clip_ratio/region_mean": 0.1931385938078165, "reward_total_mean": 0.22936499118804932, "reward_meter_mean": 0.40905338525772095, "reward_meter_std": 0.38333770632743835, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9920638799667358, "reward_repeat_soft_std": 0.010272251442074776, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.22936499118804932, "reward_total_composite_std": 0.12850287556648254} {"timestamp_utc": "2026-04-13T12:57:05Z", "mode": "train", "global_step": 95, "epoch": 0.009542943244600702, "loss": 0.2013, "grad_norm": 10.570873260498047, "learning_rate": 9.715151515151516e-06, "num_tokens": 173021.0, "completions/mean_length": 89.875, "completions/min_length": 68.0, "completions/max_length": 132.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.875, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 132.0, "rewards/meter/mean": 0.47886359691619873, "rewards/meter/std": 0.29336100816726685, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9957594275474548, "rewards/repeat_soft/std": 0.0027065465692430735, "rewards/judge_quality/mean": 0.3525000214576721, "rewards/judge_quality/std": 0.14300350844860077, "rewards/total_composite/mean": 0.24157394468784332, "rewards/total_composite/std": 0.1338033378124237, "reward": 0.24157394468784332, "reward_std": 0.1338033378124237, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23752599954605103, "sampling/sampling_logp_difference/max": 1.6505331993103027, "sampling/importance_sampling_ratio/min": 0.19194753468036652, "sampling/importance_sampling_ratio/mean": 1.061669945716858, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6323418766260147, "clip_ratio/low_mean": 0.12274500727653503, "clip_ratio/low_min": 0.12274500727653503, "clip_ratio/high_mean": 0.08438725769519806, "clip_ratio/high_max": 0.08438725769519806, "clip_ratio/region_mean": 0.2071322649717331, "reward_total_mean": 0.24157394468784332, "reward_meter_mean": 0.47886359691619873, "reward_meter_std": 0.29336100816726685, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9957594275474548, "reward_repeat_soft_std": 0.0027065465692430735, "reward_judge_quality_mean": 0.3525000214576721, "reward_judge_quality_std": 0.14300350844860077, "reward_total_composite_mean": 0.24157394468784332, "reward_total_composite_std": 0.1338033378124237} {"timestamp_utc": "2026-04-13T12:57:11Z", "mode": "train", "global_step": 96, "epoch": 0.009643395278754395, "loss": 0.0583, "grad_norm": 19.531837463378906, "learning_rate": 9.712121212121213e-06, "num_tokens": 174523.0, "completions/mean_length": 32.75, "completions/min_length": 24.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.75, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.5699098110198975, "rewards/meter/std": 0.35725337266921997, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9993458390235901, "rewards/repeat_soft/std": 0.0018502887105569243, "rewards/judge_quality/mean": 0.3062500059604645, "rewards/judge_quality/std": 0.09635314345359802, "rewards/total_composite/mean": 0.22203493118286133, "rewards/total_composite/std": 0.1043752059340477, "reward": 0.22203493118286133, "reward_std": 0.1043751984834671, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24431182444095612, "sampling/sampling_logp_difference/max": 1.7511816024780273, "sampling/importance_sampling_ratio/min": 0.1735687255859375, "sampling/importance_sampling_ratio/mean": 1.0452316999435425, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4502941220998764, "clip_ratio/low_mean": 0.12524668499827385, "clip_ratio/low_min": 0.12524668499827385, "clip_ratio/high_mean": 0.09808298293501139, "clip_ratio/high_max": 0.09808298293501139, "clip_ratio/region_mean": 0.22332966793328524, "reward_total_mean": 0.22203493118286133, "reward_meter_mean": 0.5699098110198975, "reward_meter_std": 0.35725337266921997, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9993458390235901, "reward_repeat_soft_std": 0.0018502887105569243, "reward_judge_quality_mean": 0.3062500059604645, "reward_judge_quality_std": 0.09635314345359802, "reward_total_composite_mean": 0.22203493118286133, "reward_total_composite_std": 0.1043752059340477} {"timestamp_utc": "2026-04-13T12:57:17Z", "mode": "train", "global_step": 97, "epoch": 0.009743847312908087, "loss": 0.0396, "grad_norm": 16.809978485107422, "learning_rate": 9.70909090909091e-06, "num_tokens": 176099.0, "completions/mean_length": 32.0, "completions/min_length": 26.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 32.0, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.3821866512298584, "rewards/meter/std": 0.34682658314704895, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9970651865005493, "rewards/repeat_soft/std": 0.004848456010222435, "rewards/judge_quality/mean": 0.273749977350235, "rewards/judge_quality/std": 0.09085899591445923, "rewards/total_composite/mean": 0.14850452542304993, "rewards/total_composite/std": 0.12945693731307983, "reward": 0.14850452542304993, "reward_std": 0.12945693731307983, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24849790334701538, "sampling/sampling_logp_difference/max": 1.1252381801605225, "sampling/importance_sampling_ratio/min": 0.32457512617111206, "sampling/importance_sampling_ratio/mean": 1.038015604019165, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6717825829982758, "clip_ratio/low_mean": 0.13812491949647665, "clip_ratio/low_min": 0.13812491949647665, "clip_ratio/high_mean": 0.042803031392395496, "clip_ratio/high_max": 0.042803031392395496, "clip_ratio/region_mean": 0.18092795088887215, "reward_total_mean": 0.14850452542304993, "reward_meter_mean": 0.3821866512298584, "reward_meter_std": 0.34682658314704895, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9970651865005493, "reward_repeat_soft_std": 0.004848456010222435, "reward_judge_quality_mean": 0.273749977350235, "reward_judge_quality_std": 0.09085899591445923, "reward_total_composite_mean": 0.14850452542304993, "reward_total_composite_std": 0.12945693731307983} {"timestamp_utc": "2026-04-13T12:57:24Z", "mode": "train", "global_step": 98, "epoch": 0.009844299347061778, "loss": 0.0738, "grad_norm": 12.655705451965332, "learning_rate": 9.706060606060606e-06, "num_tokens": 177756.0, "completions/mean_length": 43.125, "completions/min_length": 32.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.125, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.3984159827232361, "rewards/meter/std": 0.4022829234600067, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9977775812149048, "rewards/repeat_soft/std": 0.0038884899113327265, "rewards/judge_quality/mean": 0.2512499988079071, "rewards/judge_quality/std": 0.07219763845205307, "rewards/total_composite/mean": 0.16491495072841644, "rewards/total_composite/std": 0.1173827275633812, "reward": 0.16491495072841644, "reward_std": 0.11738273501396179, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23115411400794983, "sampling/sampling_logp_difference/max": 2.0975911617279053, "sampling/importance_sampling_ratio/min": 0.12275175750255585, "sampling/importance_sampling_ratio/mean": 1.0054800510406494, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1663263738155365, "clip_ratio/low_mean": 0.07282582484185696, "clip_ratio/low_min": 0.07282582484185696, "clip_ratio/high_mean": 0.06583761237561703, "clip_ratio/high_max": 0.06583761237561703, "clip_ratio/region_mean": 0.13866343721747398, "reward_total_mean": 0.16491495072841644, "reward_meter_mean": 0.3984159827232361, "reward_meter_std": 0.4022829234600067, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9977775812149048, "reward_repeat_soft_std": 0.0038884899113327265, "reward_judge_quality_mean": 0.2512499988079071, "reward_judge_quality_std": 0.07219763845205307, "reward_total_composite_mean": 0.16491495072841644, "reward_total_composite_std": 0.1173827275633812} {"timestamp_utc": "2026-04-13T12:57:30Z", "mode": "train", "global_step": 99, "epoch": 0.009944751381215469, "loss": 0.2834, "grad_norm": 23.216094970703125, "learning_rate": 9.703030303030305e-06, "num_tokens": 179079.0, "completions/mean_length": 21.375, "completions/min_length": 13.0, "completions/max_length": 30.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.375, "completions/min_terminated_length": 13.0, "completions/max_terminated_length": 30.0, "rewards/meter/mean": 0.3166063725948334, "rewards/meter/std": 0.41524726152420044, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.953080415725708, "rewards/repeat_soft/std": 0.018900122493505478, "rewards/judge_quality/mean": 0.4599999785423279, "rewards/judge_quality/std": 0.25712698698043823, "rewards/total_composite/mean": 0.284821093082428, "rewards/total_composite/std": 0.2717686891555786, "reward": 0.284821093082428, "reward_std": 0.2717686593532562, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22873538732528687, "sampling/sampling_logp_difference/max": 1.3408770561218262, "sampling/importance_sampling_ratio/min": 0.2616161108016968, "sampling/importance_sampling_ratio/mean": 0.9866707921028137, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9422717690467834, "clip_ratio/low_mean": 0.1328728124499321, "clip_ratio/low_min": 0.1328728124499321, "clip_ratio/high_mean": 0.0923076942563057, "clip_ratio/high_max": 0.0923076942563057, "clip_ratio/region_mean": 0.2251805067062378, "reward_total_mean": 0.284821093082428, "reward_meter_mean": 0.3166063725948334, "reward_meter_std": 0.41524726152420044, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.953080415725708, "reward_repeat_soft_std": 0.018900122493505478, "reward_judge_quality_mean": 0.4599999785423279, "reward_judge_quality_std": 0.25712698698043823, "reward_total_composite_mean": 0.284821093082428, "reward_total_composite_std": 0.2717686891555786} {"timestamp_utc": "2026-04-13T12:57:36Z", "mode": "train", "global_step": 100, "epoch": 0.010045203415369162, "loss": 0.0057, "grad_norm": 15.219758987426758, "learning_rate": 9.7e-06, "num_tokens": 180938.0, "completions/mean_length": 58.375, "completions/min_length": 48.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.375, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.3609089255332947, "rewards/meter/std": 0.27344444394111633, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9970800280570984, "rewards/repeat_soft/std": 0.0024874836672097445, "rewards/judge_quality/mean": 0.29499998688697815, "rewards/judge_quality/std": 0.1035098284482956, "rewards/total_composite/mean": 0.1632038652896881, "rewards/total_composite/std": 0.048654042184352875, "reward": 0.1632038652896881, "reward_std": 0.048654042184352875, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23344087600708008, "sampling/sampling_logp_difference/max": 1.4850358963012695, "sampling/importance_sampling_ratio/min": 0.2264942228794098, "sampling/importance_sampling_ratio/mean": 1.0198392868041992, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3165549635887146, "clip_ratio/low_mean": 0.10147346649318933, "clip_ratio/low_min": 0.10147346649318933, "clip_ratio/high_mean": 0.09966985508799553, "clip_ratio/high_max": 0.09966985508799553, "clip_ratio/region_mean": 0.20114332158118486, "reward_total_mean": 0.1632038652896881, "reward_meter_mean": 0.3609089255332947, "reward_meter_std": 0.27344444394111633, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9970800280570984, "reward_repeat_soft_std": 0.0024874836672097445, "reward_judge_quality_mean": 0.29499998688697815, "reward_judge_quality_std": 0.1035098284482956, "reward_total_composite_mean": 0.1632038652896881, "reward_total_composite_std": 0.048654042184352875} {"timestamp_utc": "2026-04-13T12:58:15Z", "mode": "eval", "global_step": 100, "epoch": 0.010045203415369162, "eval_loss": NaN, "eval_runtime": 39.4212, "eval_samples_per_second": 2.029, "eval_steps_per_second": 0.254, "eval_num_tokens": 180938.0, "eval_completions/mean_length": 60.3, "eval_completions/min_length": 26.1, "eval_completions/max_length": 115.9, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 60.3, "eval_completions/min_terminated_length": 26.1, "eval_completions/max_terminated_length": 115.9, "eval_rewards/meter/mean": 0.5333475530147552, "eval_rewards/meter/std": 0.35966198444366454, "eval_rewards/count_adherence/mean": 0.9702083349227906, "eval_rewards/count_adherence/std": 0.07405876778066159, "eval_rewards/hard_gate/mean": 0.9375, "eval_rewards/hard_gate/std": 0.1767766922712326, "eval_rewards/repeat_soft/mean": 0.9931838035583496, "eval_rewards/repeat_soft/std": 0.01113832175033167, "eval_rewards/judge_quality/mean": 0.28999999463558196, "eval_rewards/judge_quality/std": 0.09390390664339066, "eval_rewards/total_composite/mean": 0.19018368870019914, "eval_rewards/total_composite/std": 0.10621417425572872, "eval_reward": 0.19018368870019914, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.15492237582802773, "eval_sampling/sampling_logp_difference/max": 1.1971155643463134, "eval_sampling/importance_sampling_ratio/min": 0.30891612619161607, "eval_sampling/importance_sampling_ratio/mean": 1.0459434866905213, "eval_sampling/importance_sampling_ratio/max": 1.5796868562698365, "eval_entropy": 2.415260064601898, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.19018368870019914, "eval_reward_meter_mean": 0.5333475530147552, "eval_reward_meter_std": 0.35966198444366454, "eval_reward_count_adherence_mean": 0.9702083349227906, "eval_reward_count_adherence_std": 0.07405876778066159, "eval_reward_hard_gate_mean": 0.9375, "eval_reward_hard_gate_std": 0.1767766922712326, "eval_reward_repeat_soft_mean": 0.9931838035583496, "eval_reward_repeat_soft_std": 0.01113832175033167, "eval_reward_judge_quality_mean": 0.28999999463558196, "eval_reward_judge_quality_std": 0.09390390664339066, "eval_reward_total_composite_mean": 0.19018368870019914, "eval_reward_total_composite_std": 0.10621417425572872} {"timestamp_utc": "2026-04-13T12:58:25Z", "mode": "train", "global_step": 101, "epoch": 0.010145655449522853, "loss": 0.2388, "grad_norm": 11.822274208068848, "learning_rate": 9.696969696969698e-06, "num_tokens": 182977.0, "completions/mean_length": 78.875, "completions/min_length": 52.0, "completions/max_length": 139.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.875, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 139.0, "rewards/meter/mean": 0.9055176973342896, "rewards/meter/std": 0.1499919444322586, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.2357022762298584, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9967098832130432, "rewards/repeat_soft/std": 0.0029830147977918386, "rewards/judge_quality/mean": 0.29499998688697815, "rewards/judge_quality/std": 0.1035098284482956, "rewards/total_composite/mean": 0.27612704038619995, "rewards/total_composite/std": 0.11570548266172409, "reward": 0.27612704038619995, "reward_std": 0.11570549011230469, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22361187636852264, "sampling/sampling_logp_difference/max": 1.6272878646850586, "sampling/importance_sampling_ratio/min": 0.19646167755126953, "sampling/importance_sampling_ratio/mean": 1.0615099668502808, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.19254332780838, "clip_ratio/low_mean": 0.11758391372859478, "clip_ratio/low_min": 0.11758391372859478, "clip_ratio/high_mean": 0.07190426252782345, "clip_ratio/high_max": 0.07190426252782345, "clip_ratio/region_mean": 0.18948817625641823, "reward_total_mean": 0.27612704038619995, "reward_meter_mean": 0.9055176973342896, "reward_meter_std": 0.1499919444322586, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.2357022762298584, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9967098832130432, "reward_repeat_soft_std": 0.0029830147977918386, "reward_judge_quality_mean": 0.29499998688697815, "reward_judge_quality_std": 0.1035098284482956, "reward_total_composite_mean": 0.27612704038619995, "reward_total_composite_std": 0.11570548266172409} {"timestamp_utc": "2026-04-13T12:58:32Z", "mode": "train", "global_step": 102, "epoch": 0.010246107483676544, "loss": 0.0634, "grad_norm": 12.916412353515625, "learning_rate": 9.693939393939395e-06, "num_tokens": 185048.0, "completions/mean_length": 79.875, "completions/min_length": 74.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.875, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.22276419401168823, "rewards/meter/std": 0.20111523568630219, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9941090941429138, "rewards/repeat_soft/std": 0.0047340840101242065, "rewards/judge_quality/mean": 0.24250000715255737, "rewards/judge_quality/std": 0.031052952632308006, "rewards/total_composite/mean": 0.11771676689386368, "rewards/total_composite/std": 0.029656479135155678, "reward": 0.11771676689386368, "reward_std": 0.02965647727251053, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20215699076652527, "sampling/sampling_logp_difference/max": 1.4278860092163086, "sampling/importance_sampling_ratio/min": 0.239815354347229, "sampling/importance_sampling_ratio/mean": 1.0422965288162231, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8850602060556412, "clip_ratio/low_mean": 0.11844531446695328, "clip_ratio/low_min": 0.11844531446695328, "clip_ratio/high_mean": 0.056453270837664604, "clip_ratio/high_max": 0.056453270837664604, "clip_ratio/region_mean": 0.17489858530461788, "reward_total_mean": 0.11771676689386368, "reward_meter_mean": 0.22276419401168823, "reward_meter_std": 0.20111523568630219, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9941090941429138, "reward_repeat_soft_std": 0.0047340840101242065, "reward_judge_quality_mean": 0.24250000715255737, "reward_judge_quality_std": 0.031052952632308006, "reward_total_composite_mean": 0.11771676689386368, "reward_total_composite_std": 0.029656479135155678} {"timestamp_utc": "2026-04-13T12:58:39Z", "mode": "train", "global_step": 103, "epoch": 0.010346559517830235, "loss": 0.1125, "grad_norm": 17.527645111083984, "learning_rate": 9.690909090909092e-06, "num_tokens": 186673.0, "completions/mean_length": 46.125, "completions/min_length": 37.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.125, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.8049889802932739, "rewards/meter/std": 0.28971365094184875, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9981218576431274, "rewards/repeat_soft/std": 0.004964757245033979, "rewards/judge_quality/mean": 0.40625, "rewards/judge_quality/std": 0.18715444207191467, "rewards/total_composite/mean": 0.3515571355819702, "rewards/total_composite/std": 0.2247399538755417, "reward": 0.3515571355819702, "reward_std": 0.2247399389743805, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18538333475589752, "sampling/sampling_logp_difference/max": 2.0367231369018555, "sampling/importance_sampling_ratio/min": 0.13045549392700195, "sampling/importance_sampling_ratio/mean": 1.012653112411499, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.654091089963913, "clip_ratio/low_mean": 0.06139914924278855, "clip_ratio/low_min": 0.06139914924278855, "clip_ratio/high_mean": 0.11172189377248287, "clip_ratio/high_max": 0.11172189377248287, "clip_ratio/region_mean": 0.17312104301527143, "reward_total_mean": 0.3515571355819702, "reward_meter_mean": 0.8049889802932739, "reward_meter_std": 0.28971365094184875, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9981218576431274, "reward_repeat_soft_std": 0.004964757245033979, "reward_judge_quality_mean": 0.40625, "reward_judge_quality_std": 0.18715444207191467, "reward_total_composite_mean": 0.3515571355819702, "reward_total_composite_std": 0.2247399538755417} {"timestamp_utc": "2026-04-13T12:58:45Z", "mode": "train", "global_step": 104, "epoch": 0.010447011551983928, "loss": -0.0233, "grad_norm": 18.21098518371582, "learning_rate": 9.687878787878788e-06, "num_tokens": 188211.0, "completions/mean_length": 38.25, "completions/min_length": 29.0, "completions/max_length": 53.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.25, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 53.0, "rewards/meter/mean": 0.6183866858482361, "rewards/meter/std": 0.4217718541622162, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9947104454040527, "rewards/repeat_soft/std": 0.00796837080270052, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.2452881783246994, "rewards/total_composite/std": 0.10636084526777267, "reward": 0.2452881783246994, "reward_std": 0.10636083781719208, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19570432603359222, "sampling/sampling_logp_difference/max": 1.4738616943359375, "sampling/importance_sampling_ratio/min": 0.2290392965078354, "sampling/importance_sampling_ratio/mean": 1.0237973928451538, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.7581131011247635, "clip_ratio/low_mean": 0.11726908572018147, "clip_ratio/low_min": 0.11726908572018147, "clip_ratio/high_mean": 0.060981093905866146, "clip_ratio/high_max": 0.060981093905866146, "clip_ratio/region_mean": 0.1782501796260476, "reward_total_mean": 0.2452881783246994, "reward_meter_mean": 0.6183866858482361, "reward_meter_std": 0.4217718541622162, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9947104454040527, "reward_repeat_soft_std": 0.00796837080270052, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.2452881783246994, "reward_total_composite_std": 0.10636084526777267} {"timestamp_utc": "2026-04-13T12:58:52Z", "mode": "train", "global_step": 105, "epoch": 0.01054746358613762, "loss": -0.0525, "grad_norm": 13.55850887298584, "learning_rate": 9.684848484848487e-06, "num_tokens": 190277.0, "completions/mean_length": 74.25, "completions/min_length": 67.0, "completions/max_length": 110.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.25, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 110.0, "rewards/meter/mean": 0.47648465633392334, "rewards/meter/std": 0.3585143983364105, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9975783824920654, "rewards/repeat_soft/std": 0.002411383204162121, "rewards/judge_quality/mean": 0.2199999988079071, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.14505738019943237, "rewards/total_composite/std": 0.051263850182294846, "reward": 0.14505738019943237, "reward_std": 0.05126384645700455, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24152730405330658, "sampling/sampling_logp_difference/max": 1.9144554138183594, "sampling/importance_sampling_ratio/min": 0.14742209017276764, "sampling/importance_sampling_ratio/mean": 1.0519094467163086, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.5489106625318527, "clip_ratio/low_mean": 0.16279510967433453, "clip_ratio/low_min": 0.16279510967433453, "clip_ratio/high_mean": 0.08875360898673534, "clip_ratio/high_max": 0.08875360898673534, "clip_ratio/region_mean": 0.25154871866106987, "reward_total_mean": 0.14505738019943237, "reward_meter_mean": 0.47648465633392334, "reward_meter_std": 0.3585143983364105, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9975783824920654, "reward_repeat_soft_std": 0.002411383204162121, "reward_judge_quality_mean": 0.2199999988079071, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.14505738019943237, "reward_total_composite_std": 0.051263850182294846} {"timestamp_utc": "2026-04-13T12:58:58Z", "mode": "train", "global_step": 106, "epoch": 0.01064791562029131, "loss": 0.003, "grad_norm": 24.08884048461914, "learning_rate": 9.681818181818182e-06, "num_tokens": 191708.0, "completions/mean_length": 26.875, "completions/min_length": 24.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.875, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.3493647277355194, "rewards/meter/std": 0.20752744376659393, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9945707321166992, "rewards/repeat_soft/std": 0.008784092962741852, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.20193137228488922, "rewards/total_composite/std": 0.06303919106721878, "reward": 0.20193137228488922, "reward_std": 0.06303918361663818, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2212163507938385, "sampling/sampling_logp_difference/max": 2.582676887512207, "sampling/importance_sampling_ratio/min": 0.07557143270969391, "sampling/importance_sampling_ratio/mean": 1.0338249206542969, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0336359068751335, "clip_ratio/low_mean": 0.13796198274940252, "clip_ratio/low_min": 0.13796198274940252, "clip_ratio/high_mean": 0.047883063554763794, "clip_ratio/high_max": 0.047883063554763794, "clip_ratio/region_mean": 0.18584504630416632, "reward_total_mean": 0.20193137228488922, "reward_meter_mean": 0.3493647277355194, "reward_meter_std": 0.20752744376659393, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9945707321166992, "reward_repeat_soft_std": 0.008784092962741852, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.20193137228488922, "reward_total_composite_std": 0.06303919106721878} {"timestamp_utc": "2026-04-13T12:59:04Z", "mode": "train", "global_step": 107, "epoch": 0.010748367654445002, "loss": 0.171, "grad_norm": 17.395442962646484, "learning_rate": 9.67878787878788e-06, "num_tokens": 193328.0, "completions/mean_length": 39.5, "completions/min_length": 29.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.5, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.4425399899482727, "rewards/meter/std": 0.3519269824028015, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9966769814491272, "rewards/repeat_soft/std": 0.00410105474293232, "rewards/judge_quality/mean": 0.45249998569488525, "rewards/judge_quality/std": 0.24182343482971191, "rewards/total_composite/mean": 0.28937357664108276, "rewards/total_composite/std": 0.1784743368625641, "reward": 0.28937357664108276, "reward_std": 0.1784743368625641, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17070765793323517, "sampling/sampling_logp_difference/max": 1.2045235633850098, "sampling/importance_sampling_ratio/min": 0.29983481764793396, "sampling/importance_sampling_ratio/mean": 0.9965779185295105, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6595074906945229, "clip_ratio/low_mean": 0.08546209614723921, "clip_ratio/low_min": 0.08546209614723921, "clip_ratio/high_mean": 0.06928638741374016, "clip_ratio/high_max": 0.06928638741374016, "clip_ratio/region_mean": 0.15474848356097937, "reward_total_mean": 0.28937357664108276, "reward_meter_mean": 0.4425399899482727, "reward_meter_std": 0.3519269824028015, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9966769814491272, "reward_repeat_soft_std": 0.00410105474293232, "reward_judge_quality_mean": 0.45249998569488525, "reward_judge_quality_std": 0.24182343482971191, "reward_total_composite_mean": 0.28937357664108276, "reward_total_composite_std": 0.1784743368625641} {"timestamp_utc": "2026-04-13T12:59:10Z", "mode": "train", "global_step": 108, "epoch": 0.010848819688598695, "loss": 0.0939, "grad_norm": 25.62848472595215, "learning_rate": 9.675757575757577e-06, "num_tokens": 194611.0, "completions/mean_length": 25.375, "completions/min_length": 20.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.375, "completions/min_terminated_length": 20.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.37729573249816895, "rewards/meter/std": 0.37614983320236206, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9955626726150513, "rewards/repeat_soft/std": 0.005067089106887579, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.14024850726127625, "rewards/total_composite/mean": 0.26623427867889404, "rewards/total_composite/std": 0.15444695949554443, "reward": 0.26623427867889404, "reward_std": 0.15444695949554443, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.26287201046943665, "sampling/sampling_logp_difference/max": 1.492593765258789, "sampling/importance_sampling_ratio/min": 0.2247888594865799, "sampling/importance_sampling_ratio/mean": 1.0210803747177124, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9178846180438995, "clip_ratio/low_mean": 0.16352272778749466, "clip_ratio/low_min": 0.16352272778749466, "clip_ratio/high_mean": 0.08564814925193787, "clip_ratio/high_max": 0.08564814925193787, "clip_ratio/region_mean": 0.24917087703943253, "reward_total_mean": 0.26623427867889404, "reward_meter_mean": 0.37729573249816895, "reward_meter_std": 0.37614983320236206, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9955626726150513, "reward_repeat_soft_std": 0.005067089106887579, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.14024850726127625, "reward_total_composite_mean": 0.26623427867889404, "reward_total_composite_std": 0.15444695949554443} {"timestamp_utc": "2026-04-13T12:59:17Z", "mode": "train", "global_step": 109, "epoch": 0.010949271722752386, "loss": 0.0497, "grad_norm": 16.82399559020996, "learning_rate": 9.672727272727274e-06, "num_tokens": 196165.0, "completions/mean_length": 34.25, "completions/min_length": 31.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.25, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.6299466490745544, "rewards/meter/std": 0.3046122193336487, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9920294284820557, "rewards/repeat_soft/std": 0.009845644235610962, "rewards/judge_quality/mean": 0.29374998807907104, "rewards/judge_quality/std": 0.1507066786289215, "rewards/total_composite/mean": 0.20779313147068024, "rewards/total_composite/std": 0.1545531302690506, "reward": 0.20779313147068024, "reward_std": 0.1545531302690506, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.229281947016716, "sampling/sampling_logp_difference/max": 1.712673306465149, "sampling/importance_sampling_ratio/min": 0.18038292229175568, "sampling/importance_sampling_ratio/mean": 1.0576488971710205, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3824533075094223, "clip_ratio/low_mean": 0.13064675964415073, "clip_ratio/low_min": 0.13064675964415073, "clip_ratio/high_mean": 0.08631171379238367, "clip_ratio/high_max": 0.08631171379238367, "clip_ratio/region_mean": 0.2169584734365344, "reward_total_mean": 0.20779313147068024, "reward_meter_mean": 0.6299466490745544, "reward_meter_std": 0.3046122193336487, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9920294284820557, "reward_repeat_soft_std": 0.009845644235610962, "reward_judge_quality_mean": 0.29374998807907104, "reward_judge_quality_std": 0.1507066786289215, "reward_total_composite_mean": 0.20779313147068024, "reward_total_composite_std": 0.1545531302690506} {"timestamp_utc": "2026-04-13T12:59:23Z", "mode": "train", "global_step": 110, "epoch": 0.011049723756906077, "loss": 0.0266, "grad_norm": 15.715481758117676, "learning_rate": 9.66969696969697e-06, "num_tokens": 197812.0, "completions/mean_length": 36.875, "completions/min_length": 35.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.875, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.9310295581817627, "rewards/meter/std": 0.1262810379266739, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9877753257751465, "rewards/repeat_soft/std": 0.014998417347669601, "rewards/judge_quality/mean": 0.4987499713897705, "rewards/judge_quality/std": 0.23277442157268524, "rewards/total_composite/mean": 0.48087847232818604, "rewards/total_composite/std": 0.2299085557460785, "reward": 0.48087847232818604, "reward_std": 0.2299085408449173, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19083642959594727, "sampling/sampling_logp_difference/max": 1.0480821132659912, "sampling/importance_sampling_ratio/min": 0.3924664855003357, "sampling/importance_sampling_ratio/mean": 1.069949984550476, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.354040265083313, "clip_ratio/low_mean": 0.10091430973261595, "clip_ratio/low_min": 0.10091430973261595, "clip_ratio/high_mean": 0.04637694172561169, "clip_ratio/high_max": 0.04637694172561169, "clip_ratio/region_mean": 0.14729125145822763, "reward_total_mean": 0.48087847232818604, "reward_meter_mean": 0.9310295581817627, "reward_meter_std": 0.1262810379266739, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9877753257751465, "reward_repeat_soft_std": 0.014998417347669601, "reward_judge_quality_mean": 0.4987499713897705, "reward_judge_quality_std": 0.23277442157268524, "reward_total_composite_mean": 0.48087847232818604, "reward_total_composite_std": 0.2299085557460785} {"timestamp_utc": "2026-04-13T12:59:30Z", "mode": "train", "global_step": 111, "epoch": 0.01115017579105977, "loss": 0.0448, "grad_norm": 17.277462005615234, "learning_rate": 9.666666666666667e-06, "num_tokens": 199647.0, "completions/mean_length": 65.375, "completions/min_length": 54.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.375, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.5294692516326904, "rewards/meter/std": 0.29245197772979736, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9982734322547913, "rewards/repeat_soft/std": 0.0013606828870251775, "rewards/judge_quality/mean": 0.3024999797344208, "rewards/judge_quality/std": 0.09939099848270416, "rewards/total_composite/mean": 0.20176908373832703, "rewards/total_composite/std": 0.11734206974506378, "reward": 0.20176908373832703, "reward_std": 0.11734206229448318, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23416373133659363, "sampling/sampling_logp_difference/max": 1.5392494201660156, "sampling/importance_sampling_ratio/min": 0.21454209089279175, "sampling/importance_sampling_ratio/mean": 1.045576572418213, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.953499972820282, "clip_ratio/low_mean": 0.07657067384570837, "clip_ratio/low_min": 0.07657067384570837, "clip_ratio/high_mean": 0.1422189585864544, "clip_ratio/high_max": 0.1422189585864544, "clip_ratio/region_mean": 0.21878963243216276, "reward_total_mean": 0.20176908373832703, "reward_meter_mean": 0.5294692516326904, "reward_meter_std": 0.29245197772979736, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9982734322547913, "reward_repeat_soft_std": 0.0013606828870251775, "reward_judge_quality_mean": 0.3024999797344208, "reward_judge_quality_std": 0.09939099848270416, "reward_total_composite_mean": 0.20176908373832703, "reward_total_composite_std": 0.11734206974506378} {"timestamp_utc": "2026-04-13T12:59:36Z", "mode": "train", "global_step": 112, "epoch": 0.011250627825213461, "loss": -0.0488, "grad_norm": 26.42018699645996, "learning_rate": 9.663636363636364e-06, "num_tokens": 201110.0, "completions/mean_length": 17.875, "completions/min_length": 14.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 17.875, "completions/min_terminated_length": 14.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.5070533156394958, "rewards/meter/std": 0.4634113907814026, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.6875, "rewards/judge_quality/std": 0.36228442192077637, "rewards/total_composite/mean": 0.44090500473976135, "rewards/total_composite/std": 0.31895413994789124, "reward": 0.44090500473976135, "reward_std": 0.31895413994789124, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17994388937950134, "sampling/sampling_logp_difference/max": 1.7454051971435547, "sampling/importance_sampling_ratio/min": 0.17457422614097595, "sampling/importance_sampling_ratio/mean": 1.0285303592681885, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4367366135120392, "clip_ratio/low_mean": 0.11488095764070749, "clip_ratio/low_min": 0.11488095764070749, "clip_ratio/high_mean": 0.01666666753590107, "clip_ratio/high_max": 0.01666666753590107, "clip_ratio/region_mean": 0.13154762517660856, "reward_total_mean": 0.44090500473976135, "reward_meter_mean": 0.5070533156394958, "reward_meter_std": 0.4634113907814026, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.6875, "reward_judge_quality_std": 0.36228442192077637, "reward_total_composite_mean": 0.44090500473976135, "reward_total_composite_std": 0.31895413994789124} {"timestamp_utc": "2026-04-13T12:59:42Z", "mode": "train", "global_step": 113, "epoch": 0.011351079859367152, "loss": -0.0158, "grad_norm": 14.468746185302734, "learning_rate": 9.660606060606061e-06, "num_tokens": 202798.0, "completions/mean_length": 55.0, "completions/min_length": 46.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.0, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.2444886863231659, "rewards/meter/std": 0.281507283449173, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9952455759048462, "rewards/repeat_soft/std": 0.0036034355871379375, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.2020210474729538, "rewards/total_composite/mean": 0.20411249995231628, "rewards/total_composite/std": 0.08018417656421661, "reward": 0.20411249995231628, "reward_std": 0.08018417656421661, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1938779354095459, "sampling/sampling_logp_difference/max": 2.061924934387207, "sampling/importance_sampling_ratio/min": 0.1272088587284088, "sampling/importance_sampling_ratio/mean": 1.0262696743011475, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8172344714403152, "clip_ratio/low_mean": 0.09356675297021866, "clip_ratio/low_min": 0.09356675297021866, "clip_ratio/high_mean": 0.11305304430425167, "clip_ratio/high_max": 0.11305304430425167, "clip_ratio/region_mean": 0.20661979727447033, "reward_total_mean": 0.20411249995231628, "reward_meter_mean": 0.2444886863231659, "reward_meter_std": 0.281507283449173, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9952455759048462, "reward_repeat_soft_std": 0.0036034355871379375, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.2020210474729538, "reward_total_composite_mean": 0.20411249995231628, "reward_total_composite_std": 0.08018417656421661} {"timestamp_utc": "2026-04-13T12:59:49Z", "mode": "train", "global_step": 114, "epoch": 0.011451531893520843, "loss": -0.0393, "grad_norm": 15.590677261352539, "learning_rate": 9.657575757575758e-06, "num_tokens": 204317.0, "completions/mean_length": 35.875, "completions/min_length": 30.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.875, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.9072677493095398, "rewards/meter/std": 0.18171843886375427, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9994348883628845, "rewards/repeat_soft/std": 0.0014808415435254574, "rewards/judge_quality/mean": 0.29499998688697815, "rewards/judge_quality/std": 0.1035098284482956, "rewards/total_composite/mean": 0.28085845708847046, "rewards/total_composite/std": 0.11390554904937744, "reward": 0.28085845708847046, "reward_std": 0.11390554159879684, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2342054545879364, "sampling/sampling_logp_difference/max": 2.456653594970703, "sampling/importance_sampling_ratio/min": 0.08572132885456085, "sampling/importance_sampling_ratio/mean": 1.0775035619735718, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.0509612560272217, "clip_ratio/low_mean": 0.10094697307795286, "clip_ratio/low_min": 0.10094697307795286, "clip_ratio/high_mean": 0.07347696367651224, "clip_ratio/high_max": 0.07347696367651224, "clip_ratio/region_mean": 0.1744239367544651, "reward_total_mean": 0.28085845708847046, "reward_meter_mean": 0.9072677493095398, "reward_meter_std": 0.18171843886375427, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9994348883628845, "reward_repeat_soft_std": 0.0014808415435254574, "reward_judge_quality_mean": 0.29499998688697815, "reward_judge_quality_std": 0.1035098284482956, "reward_total_composite_mean": 0.28085845708847046, "reward_total_composite_std": 0.11390554904937744} {"timestamp_utc": "2026-04-13T12:59:55Z", "mode": "train", "global_step": 115, "epoch": 0.011551983927674536, "loss": -0.0278, "grad_norm": 13.686504364013672, "learning_rate": 9.654545454545456e-06, "num_tokens": 206390.0, "completions/mean_length": 71.125, "completions/min_length": 64.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.125, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.4500429630279541, "rewards/meter/std": 0.3327544927597046, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9969667196273804, "rewards/repeat_soft/std": 0.0027483219746500254, "rewards/judge_quality/mean": 0.22749999165534973, "rewards/judge_quality/std": 0.02121320553123951, "rewards/total_composite/mean": 0.14654693007469177, "rewards/total_composite/std": 0.05158424749970436, "reward": 0.14654693007469177, "reward_std": 0.05158424377441406, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.226264089345932, "sampling/sampling_logp_difference/max": 1.7206354141235352, "sampling/importance_sampling_ratio/min": 0.1789523959159851, "sampling/importance_sampling_ratio/mean": 1.0400731563568115, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.294025242328644, "clip_ratio/low_mean": 0.07087157759815454, "clip_ratio/low_min": 0.07087157759815454, "clip_ratio/high_mean": 0.11259152367711067, "clip_ratio/high_max": 0.11259152367711067, "clip_ratio/region_mean": 0.18346310127526522, "reward_total_mean": 0.14654693007469177, "reward_meter_mean": 0.4500429630279541, "reward_meter_std": 0.3327544927597046, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9969667196273804, "reward_repeat_soft_std": 0.0027483219746500254, "reward_judge_quality_mean": 0.22749999165534973, "reward_judge_quality_std": 0.02121320553123951, "reward_total_composite_mean": 0.14654693007469177, "reward_total_composite_std": 0.05158424749970436} {"timestamp_utc": "2026-04-13T13:00:01Z", "mode": "train", "global_step": 116, "epoch": 0.011652435961828227, "loss": -0.0171, "grad_norm": 16.382169723510742, "learning_rate": 9.651515151515153e-06, "num_tokens": 207902.0, "completions/mean_length": 34.0, "completions/min_length": 31.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.0, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.8620812892913818, "rewards/meter/std": 0.3167654573917389, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9956598281860352, "rewards/repeat_soft/std": 0.009578917175531387, "rewards/judge_quality/mean": 0.44874995946884155, "rewards/judge_quality/std": 0.12088219076395035, "rewards/total_composite/mean": 0.41245341300964355, "rewards/total_composite/std": 0.15856164693832397, "reward": 0.41245341300964355, "reward_std": 0.15856164693832397, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.188541442155838, "sampling/sampling_logp_difference/max": 1.8721799850463867, "sampling/importance_sampling_ratio/min": 0.15378804504871368, "sampling/importance_sampling_ratio/mean": 1.039787769317627, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6071799099445343, "clip_ratio/low_mean": 0.04992884211242199, "clip_ratio/low_min": 0.04992884211242199, "clip_ratio/high_mean": 0.1412046840414405, "clip_ratio/high_max": 0.1412046840414405, "clip_ratio/region_mean": 0.19113352615386248, "reward_total_mean": 0.41245341300964355, "reward_meter_mean": 0.8620812892913818, "reward_meter_std": 0.3167654573917389, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9956598281860352, "reward_repeat_soft_std": 0.009578917175531387, "reward_judge_quality_mean": 0.44874995946884155, "reward_judge_quality_std": 0.12088219076395035, "reward_total_composite_mean": 0.41245341300964355, "reward_total_composite_std": 0.15856164693832397} {"timestamp_utc": "2026-04-13T13:00:08Z", "mode": "train", "global_step": 117, "epoch": 0.011752887995981919, "loss": 0.0325, "grad_norm": 17.374835968017578, "learning_rate": 9.648484848484849e-06, "num_tokens": 209420.0, "completions/mean_length": 38.75, "completions/min_length": 36.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.75, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.7884187698364258, "rewards/meter/std": 0.2108745276927948, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9941059947013855, "rewards/repeat_soft/std": 0.006391895469278097, "rewards/judge_quality/mean": 0.23499999940395355, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.2026297152042389, "rewards/total_composite/std": 0.03590220957994461, "reward": 0.2026297152042389, "reward_std": 0.03590220585465431, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09602572023868561, "sampling/sampling_logp_difference/max": 1.2067923545837402, "sampling/importance_sampling_ratio/min": 0.2991553246974945, "sampling/importance_sampling_ratio/mean": 0.9958279132843018, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.48540198802948, "clip_ratio/low_mean": 0.05902969278395176, "clip_ratio/low_min": 0.05902969278395176, "clip_ratio/high_mean": 0.043296586256474257, "clip_ratio/high_max": 0.043296586256474257, "clip_ratio/region_mean": 0.10232627904042602, "reward_total_mean": 0.2026297152042389, "reward_meter_mean": 0.7884187698364258, "reward_meter_std": 0.2108745276927948, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9941059947013855, "reward_repeat_soft_std": 0.006391895469278097, "reward_judge_quality_mean": 0.23499999940395355, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.2026297152042389, "reward_total_composite_std": 0.03590220957994461} {"timestamp_utc": "2026-04-13T13:00:15Z", "mode": "train", "global_step": 118, "epoch": 0.01185334003013561, "loss": 0.0367, "grad_norm": 10.21153450012207, "learning_rate": 9.645454545454548e-06, "num_tokens": 211862.0, "completions/mean_length": 105.25, "completions/min_length": 90.0, "completions/max_length": 134.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 105.25, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 134.0, "rewards/meter/mean": 0.6786713600158691, "rewards/meter/std": 0.38632097840309143, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9965351223945618, "rewards/repeat_soft/std": 0.004216012079268694, "rewards/judge_quality/mean": 0.26999998092651367, "rewards/judge_quality/std": 0.06989788264036179, "rewards/total_composite/mean": 0.19293062388896942, "rewards/total_composite/std": 0.1444544941186905, "reward": 0.19293062388896942, "reward_std": 0.1444544941186905, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24396754801273346, "sampling/sampling_logp_difference/max": 2.569154739379883, "sampling/importance_sampling_ratio/min": 0.07660026848316193, "sampling/importance_sampling_ratio/mean": 1.0612008571624756, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.2306472063064575, "clip_ratio/low_mean": 0.0684139784425497, "clip_ratio/low_min": 0.0684139784425497, "clip_ratio/high_mean": 0.1381640676409006, "clip_ratio/high_max": 0.1381640676409006, "clip_ratio/region_mean": 0.20657804608345032, "reward_total_mean": 0.19293062388896942, "reward_meter_mean": 0.6786713600158691, "reward_meter_std": 0.38632097840309143, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9965351223945618, "reward_repeat_soft_std": 0.004216012079268694, "reward_judge_quality_mean": 0.26999998092651367, "reward_judge_quality_std": 0.06989788264036179, "reward_total_composite_mean": 0.19293062388896942, "reward_total_composite_std": 0.1444544941186905} {"timestamp_utc": "2026-04-13T13:00:22Z", "mode": "train", "global_step": 119, "epoch": 0.011953792064289303, "loss": 0.171, "grad_norm": 15.081986427307129, "learning_rate": 9.642424242424243e-06, "num_tokens": 213532.0, "completions/mean_length": 42.75, "completions/min_length": 33.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.75, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.39194825291633606, "rewards/meter/std": 0.38216766715049744, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9988094568252563, "rewards/repeat_soft/std": 0.001973578706383705, "rewards/judge_quality/mean": 0.26375001668930054, "rewards/judge_quality/std": 0.0645727664232254, "rewards/total_composite/mean": 0.15987950563430786, "rewards/total_composite/std": 0.07312759011983871, "reward": 0.15987950563430786, "reward_std": 0.07312759757041931, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2222803384065628, "sampling/sampling_logp_difference/max": 1.1502761840820312, "sampling/importance_sampling_ratio/min": 0.3165493309497833, "sampling/importance_sampling_ratio/mean": 1.0256190299987793, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.469740629196167, "clip_ratio/low_mean": 0.11986222676932812, "clip_ratio/low_min": 0.11986222676932812, "clip_ratio/high_mean": 0.12315725162625313, "clip_ratio/high_max": 0.12315725162625313, "clip_ratio/region_mean": 0.24301947839558125, "reward_total_mean": 0.15987950563430786, "reward_meter_mean": 0.39194825291633606, "reward_meter_std": 0.38216766715049744, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9988094568252563, "reward_repeat_soft_std": 0.001973578706383705, "reward_judge_quality_mean": 0.26375001668930054, "reward_judge_quality_std": 0.0645727664232254, "reward_total_composite_mean": 0.15987950563430786, "reward_total_composite_std": 0.07312759011983871} {"timestamp_utc": "2026-04-13T13:00:30Z", "mode": "train", "global_step": 120, "epoch": 0.012054244098442994, "loss": 0.2572, "grad_norm": 16.60124969482422, "learning_rate": 9.63939393939394e-06, "num_tokens": 215614.0, "completions/mean_length": 66.25, "completions/min_length": 49.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.25, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.6359738111495972, "rewards/meter/std": 0.3151750862598419, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9953513741493225, "rewards/repeat_soft/std": 0.003892518114298582, "rewards/judge_quality/mean": 0.32499998807907104, "rewards/judge_quality/std": 0.08124037832021713, "rewards/total_composite/mean": 0.25074681639671326, "rewards/total_composite/std": 0.107984259724617, "reward": 0.25074681639671326, "reward_std": 0.107984259724617, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24169640243053436, "sampling/sampling_logp_difference/max": 1.6041224002838135, "sampling/importance_sampling_ratio/min": 0.20106592774391174, "sampling/importance_sampling_ratio/mean": 1.0396535396575928, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.5456028133630753, "clip_ratio/low_mean": 0.11237266473472118, "clip_ratio/low_min": 0.11237266473472118, "clip_ratio/high_mean": 0.112040089443326, "clip_ratio/high_max": 0.112040089443326, "clip_ratio/region_mean": 0.22441275417804718, "reward_total_mean": 0.25074681639671326, "reward_meter_mean": 0.6359738111495972, "reward_meter_std": 0.3151750862598419, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9953513741493225, "reward_repeat_soft_std": 0.003892518114298582, "reward_judge_quality_mean": 0.32499998807907104, "reward_judge_quality_std": 0.08124037832021713, "reward_total_composite_mean": 0.25074681639671326, "reward_total_composite_std": 0.107984259724617} {"timestamp_utc": "2026-04-13T13:00:38Z", "mode": "train", "global_step": 121, "epoch": 0.012154696132596685, "loss": -0.0114, "grad_norm": 10.472457885742188, "learning_rate": 9.636363636363638e-06, "num_tokens": 218108.0, "completions/mean_length": 111.75, "completions/min_length": 93.0, "completions/max_length": 149.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 111.75, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 149.0, "rewards/meter/mean": 0.21533173322677612, "rewards/meter/std": 0.13730649650096893, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9945930242538452, "rewards/repeat_soft/std": 0.004704488907009363, "rewards/judge_quality/mean": 0.2199999988079071, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.10761401057243347, "rewards/total_composite/std": 0.019664164632558823, "reward": 0.10761401057243347, "reward_std": 0.019664164632558823, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2531386911869049, "sampling/sampling_logp_difference/max": 1.755446434020996, "sampling/importance_sampling_ratio/min": 0.17283006012439728, "sampling/importance_sampling_ratio/mean": 1.0559929609298706, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.2308313846588135, "clip_ratio/low_mean": 0.0991344191133976, "clip_ratio/low_min": 0.0991344191133976, "clip_ratio/high_mean": 0.10845869779586792, "clip_ratio/high_max": 0.10845869779586792, "clip_ratio/region_mean": 0.20759311690926552, "reward_total_mean": 0.10761401057243347, "reward_meter_mean": 0.21533173322677612, "reward_meter_std": 0.13730649650096893, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9945930242538452, "reward_repeat_soft_std": 0.004704488907009363, "reward_judge_quality_mean": 0.2199999988079071, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.10761401057243347, "reward_total_composite_std": 0.019664164632558823} {"timestamp_utc": "2026-04-13T13:00:45Z", "mode": "train", "global_step": 122, "epoch": 0.012255148166750376, "loss": 0.0792, "grad_norm": 16.2755184173584, "learning_rate": 9.633333333333335e-06, "num_tokens": 219713.0, "completions/mean_length": 42.625, "completions/min_length": 34.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.625, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.7699140906333923, "rewards/meter/std": 0.3879815936088562, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9996742010116577, "rewards/repeat_soft/std": 0.0009215619065798819, "rewards/judge_quality/mean": 0.3449999988079071, "rewards/judge_quality/std": 0.14880476891994476, "rewards/total_composite/mean": 0.28357014060020447, "rewards/total_composite/std": 0.19001367688179016, "reward": 0.28357014060020447, "reward_std": 0.19001366198062897, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23517604172229767, "sampling/sampling_logp_difference/max": 1.3021726608276367, "sampling/importance_sampling_ratio/min": 0.27194032073020935, "sampling/importance_sampling_ratio/mean": 1.0255141258239746, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.5738581717014313, "clip_ratio/low_mean": 0.09402434760704637, "clip_ratio/low_min": 0.09402434760704637, "clip_ratio/high_mean": 0.07149197906255722, "clip_ratio/high_max": 0.07149197906255722, "clip_ratio/region_mean": 0.1655163266696036, "reward_total_mean": 0.28357014060020447, "reward_meter_mean": 0.7699140906333923, "reward_meter_std": 0.3879815936088562, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9996742010116577, "reward_repeat_soft_std": 0.0009215619065798819, "reward_judge_quality_mean": 0.3449999988079071, "reward_judge_quality_std": 0.14880476891994476, "reward_total_composite_mean": 0.28357014060020447, "reward_total_composite_std": 0.19001367688179016} {"timestamp_utc": "2026-04-13T13:00:53Z", "mode": "train", "global_step": 123, "epoch": 0.012355600200904069, "loss": 0.0079, "grad_norm": 25.827007293701172, "learning_rate": 9.63030303030303e-06, "num_tokens": 221099.0, "completions/mean_length": 25.25, "completions/min_length": 23.0, "completions/max_length": 29.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.25, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 29.0, "rewards/meter/mean": 0.48179739713668823, "rewards/meter/std": 0.37739667296409607, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9988503456115723, "rewards/repeat_soft/std": 0.002489580772817135, "rewards/judge_quality/mean": 0.4987499713897705, "rewards/judge_quality/std": 0.1769937425851822, "rewards/total_composite/mean": 0.338995099067688, "rewards/total_composite/std": 0.18322403728961945, "reward": 0.338995099067688, "reward_std": 0.18322403728961945, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2206656038761139, "sampling/sampling_logp_difference/max": 2.0103414058685303, "sampling/importance_sampling_ratio/min": 0.13394293189048767, "sampling/importance_sampling_ratio/mean": 1.0055018663406372, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.435492642223835, "clip_ratio/low_mean": 0.08452132297679782, "clip_ratio/low_min": 0.08452132297679782, "clip_ratio/high_mean": 0.07278680521994829, "clip_ratio/high_max": 0.07278680521994829, "clip_ratio/region_mean": 0.1573081281967461, "reward_total_mean": 0.338995099067688, "reward_meter_mean": 0.48179739713668823, "reward_meter_std": 0.37739667296409607, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9988503456115723, "reward_repeat_soft_std": 0.002489580772817135, "reward_judge_quality_mean": 0.4987499713897705, "reward_judge_quality_std": 0.1769937425851822, "reward_total_composite_mean": 0.338995099067688, "reward_total_composite_std": 0.18322403728961945} {"timestamp_utc": "2026-04-13T13:00:59Z", "mode": "train", "global_step": 124, "epoch": 0.01245605223505776, "loss": -0.0173, "grad_norm": 23.003311157226562, "learning_rate": 9.627272727272728e-06, "num_tokens": 222427.0, "completions/mean_length": 17.0, "completions/min_length": 13.0, "completions/max_length": 23.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 17.0, "completions/min_terminated_length": 13.0, "completions/max_terminated_length": 23.0, "rewards/meter/mean": 0.7263355255126953, "rewards/meter/std": 0.4465133249759674, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.961837112903595, "rewards/repeat_soft/std": 0.0018749026348814368, "rewards/judge_quality/mean": 0.3387499749660492, "rewards/judge_quality/std": 0.09538455307483673, "rewards/total_composite/mean": 0.2906777858734131, "rewards/total_composite/std": 0.14767445623874664, "reward": 0.2906777858734131, "reward_std": 0.14767445623874664, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18750518560409546, "sampling/sampling_logp_difference/max": 0.9341144561767578, "sampling/importance_sampling_ratio/min": 0.3929336667060852, "sampling/importance_sampling_ratio/mean": 1.0421078205108643, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8559207618236542, "clip_ratio/low_mean": 0.06732460018247366, "clip_ratio/low_min": 0.06732460018247366, "clip_ratio/high_mean": 0.12367216311395168, "clip_ratio/high_max": 0.12367216311395168, "clip_ratio/region_mean": 0.19099676329642534, "reward_total_mean": 0.2906777858734131, "reward_meter_mean": 0.7263355255126953, "reward_meter_std": 0.4465133249759674, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.961837112903595, "reward_repeat_soft_std": 0.0018749026348814368, "reward_judge_quality_mean": 0.3387499749660492, "reward_judge_quality_std": 0.09538455307483673, "reward_total_composite_mean": 0.2906777858734131, "reward_total_composite_std": 0.14767445623874664} {"timestamp_utc": "2026-04-13T13:01:06Z", "mode": "train", "global_step": 125, "epoch": 0.012556504269211451, "loss": 0.0823, "grad_norm": 19.11264991760254, "learning_rate": 9.624242424242425e-06, "num_tokens": 224245.0, "completions/mean_length": 58.25, "completions/min_length": 49.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.25, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.3744819164276123, "rewards/meter/std": 0.2976786494255066, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9952961206436157, "rewards/repeat_soft/std": 0.0038182511925697327, "rewards/judge_quality/mean": 0.3100000023841858, "rewards/judge_quality/std": 0.14262838661670685, "rewards/total_composite/mean": 0.17342031002044678, "rewards/total_composite/std": 0.0660971999168396, "reward": 0.17342031002044678, "reward_std": 0.0660971999168396, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2614648938179016, "sampling/sampling_logp_difference/max": 2.5867881774902344, "sampling/importance_sampling_ratio/min": 0.07526137679815292, "sampling/importance_sampling_ratio/mean": 1.0358235836029053, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.239851325750351, "clip_ratio/low_mean": 0.1010020449757576, "clip_ratio/low_min": 0.1010020449757576, "clip_ratio/high_mean": 0.12148526310920715, "clip_ratio/high_max": 0.12148526310920715, "clip_ratio/region_mean": 0.22248730808496475, "reward_total_mean": 0.17342031002044678, "reward_meter_mean": 0.3744819164276123, "reward_meter_std": 0.2976786494255066, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9952961206436157, "reward_repeat_soft_std": 0.0038182511925697327, "reward_judge_quality_mean": 0.3100000023841858, "reward_judge_quality_std": 0.14262838661670685, "reward_total_composite_mean": 0.17342031002044678, "reward_total_composite_std": 0.0660971999168396} {"timestamp_utc": "2026-04-13T13:01:15Z", "mode": "train", "global_step": 126, "epoch": 0.012656956303365142, "loss": 0.2376, "grad_norm": 19.68045425415039, "learning_rate": 9.621212121212122e-06, "num_tokens": 225588.0, "completions/mean_length": 23.875, "completions/min_length": 16.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.875, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.3367393910884857, "rewards/meter/std": 0.44271981716156006, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9657489061355591, "rewards/repeat_soft/std": 0.014413577504456043, "rewards/judge_quality/mean": 0.22624999284744263, "rewards/judge_quality/std": 0.02386719174683094, "rewards/total_composite/mean": 0.12879985570907593, "rewards/total_composite/std": 0.06986033171415329, "reward": 0.12879985570907593, "reward_std": 0.06986033171415329, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2064133882522583, "sampling/sampling_logp_difference/max": 1.1316213607788086, "sampling/importance_sampling_ratio/min": 0.3225099444389343, "sampling/importance_sampling_ratio/mean": 1.0287368297576904, "sampling/importance_sampling_ratio/max": 1.9636597633361816, "entropy": 2.439403772354126, "clip_ratio/low_mean": 0.1330265123397112, "clip_ratio/low_min": 0.1330265123397112, "clip_ratio/high_mean": 0.07791610900312662, "clip_ratio/high_max": 0.07791610900312662, "clip_ratio/region_mean": 0.2109426213428378, "reward_total_mean": 0.12879985570907593, "reward_meter_mean": 0.3367393910884857, "reward_meter_std": 0.44271981716156006, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9657489061355591, "reward_repeat_soft_std": 0.014413577504456043, "reward_judge_quality_mean": 0.22624999284744263, "reward_judge_quality_std": 0.02386719174683094, "reward_total_composite_mean": 0.12879985570907593, "reward_total_composite_std": 0.06986033171415329} {"timestamp_utc": "2026-04-13T13:01:22Z", "mode": "train", "global_step": 127, "epoch": 0.012757408337518835, "loss": 0.0518, "grad_norm": 21.802650451660156, "learning_rate": 9.61818181818182e-06, "num_tokens": 227064.0, "completions/mean_length": 38.5, "completions/min_length": 31.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.5, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.6797225475311279, "rewards/meter/std": 0.3927367329597473, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9996022582054138, "rewards/repeat_soft/std": 0.001124941511079669, "rewards/judge_quality/mean": 0.26375001668930054, "rewards/judge_quality/std": 0.0645727664232254, "rewards/total_composite/mean": 0.2068263292312622, "rewards/total_composite/std": 0.06877505034208298, "reward": 0.2068263292312622, "reward_std": 0.06877505779266357, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23567445576190948, "sampling/sampling_logp_difference/max": 1.6988935470581055, "sampling/importance_sampling_ratio/min": 0.1828857660293579, "sampling/importance_sampling_ratio/mean": 1.067500114440918, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.59160253405571, "clip_ratio/low_mean": 0.04095253814011812, "clip_ratio/low_min": 0.04095253814011812, "clip_ratio/high_mean": 0.15336354076862335, "clip_ratio/high_max": 0.15336354076862335, "clip_ratio/region_mean": 0.19431607890874147, "reward_total_mean": 0.2068263292312622, "reward_meter_mean": 0.6797225475311279, "reward_meter_std": 0.3927367329597473, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9996022582054138, "reward_repeat_soft_std": 0.001124941511079669, "reward_judge_quality_mean": 0.26375001668930054, "reward_judge_quality_std": 0.0645727664232254, "reward_total_composite_mean": 0.2068263292312622, "reward_total_composite_std": 0.06877505034208298} {"timestamp_utc": "2026-04-13T13:01:29Z", "mode": "train", "global_step": 128, "epoch": 0.012857860371672527, "loss": 0.0646, "grad_norm": 13.482661247253418, "learning_rate": 9.615151515151517e-06, "num_tokens": 228989.0, "completions/mean_length": 67.625, "completions/min_length": 58.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.625, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.3872728943824768, "rewards/meter/std": 0.32985642552375793, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9928202629089355, "rewards/repeat_soft/std": 0.006111267488449812, "rewards/judge_quality/mean": 0.3199999928474426, "rewards/judge_quality/std": 0.18516401946544647, "rewards/total_composite/mean": 0.19855639338493347, "rewards/total_composite/std": 0.15774224698543549, "reward": 0.19855639338493347, "reward_std": 0.15774224698543549, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22792702913284302, "sampling/sampling_logp_difference/max": 1.5525579452514648, "sampling/importance_sampling_ratio/min": 0.21170574426651, "sampling/importance_sampling_ratio/mean": 1.0355592966079712, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.207075983285904, "clip_ratio/low_mean": 0.11547990329563618, "clip_ratio/low_min": 0.11547990329563618, "clip_ratio/high_mean": 0.08882211707532406, "clip_ratio/high_max": 0.08882211707532406, "clip_ratio/region_mean": 0.20430202037096024, "reward_total_mean": 0.19855639338493347, "reward_meter_mean": 0.3872728943824768, "reward_meter_std": 0.32985642552375793, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9928202629089355, "reward_repeat_soft_std": 0.006111267488449812, "reward_judge_quality_mean": 0.3199999928474426, "reward_judge_quality_std": 0.18516401946544647, "reward_total_composite_mean": 0.19855639338493347, "reward_total_composite_std": 0.15774224698543549} {"timestamp_utc": "2026-04-13T13:01:35Z", "mode": "train", "global_step": 129, "epoch": 0.012958312405826218, "loss": 0.0638, "grad_norm": 10.612187385559082, "learning_rate": 9.612121212121212e-06, "num_tokens": 230746.0, "completions/mean_length": 66.625, "completions/min_length": 61.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.625, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.5210334062576294, "rewards/meter/std": 0.3286823332309723, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9986972808837891, "rewards/repeat_soft/std": 0.0018164877546951175, "rewards/judge_quality/mean": 0.25, "rewards/judge_quality/std": 0.03207135200500488, "rewards/total_composite/mean": 0.17536430060863495, "rewards/total_composite/std": 0.06717904657125473, "reward": 0.17536430060863495, "reward_std": 0.06717904657125473, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22001968324184418, "sampling/sampling_logp_difference/max": 1.4705123901367188, "sampling/importance_sampling_ratio/min": 0.22980770468711853, "sampling/importance_sampling_ratio/mean": 1.0479635000228882, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.956223636865616, "clip_ratio/low_mean": 0.10533072054386139, "clip_ratio/low_min": 0.10533072054386139, "clip_ratio/high_mean": 0.10481791384518147, "clip_ratio/high_max": 0.10481791384518147, "clip_ratio/region_mean": 0.21014863438904285, "reward_total_mean": 0.17536430060863495, "reward_meter_mean": 0.5210334062576294, "reward_meter_std": 0.3286823332309723, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9986972808837891, "reward_repeat_soft_std": 0.0018164877546951175, "reward_judge_quality_mean": 0.25, "reward_judge_quality_std": 0.03207135200500488, "reward_total_composite_mean": 0.17536430060863495, "reward_total_composite_std": 0.06717904657125473} {"timestamp_utc": "2026-04-13T13:01:42Z", "mode": "train", "global_step": 130, "epoch": 0.013058764439979909, "loss": -0.0281, "grad_norm": 11.608769416809082, "learning_rate": 9.60909090909091e-06, "num_tokens": 232821.0, "completions/mean_length": 82.375, "completions/min_length": 71.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.375, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.3977913558483124, "rewards/meter/std": 0.32337522506713867, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9964189529418945, "rewards/repeat_soft/std": 0.0026952349580824375, "rewards/judge_quality/mean": 0.22749999165534973, "rewards/judge_quality/std": 0.02121320553123951, "rewards/total_composite/mean": 0.13609255850315094, "rewards/total_composite/std": 0.04253906011581421, "reward": 0.13609255850315094, "reward_std": 0.04253906011581421, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24335147440433502, "sampling/sampling_logp_difference/max": 2.166168212890625, "sampling/importance_sampling_ratio/min": 0.11461596190929413, "sampling/importance_sampling_ratio/mean": 1.0427734851837158, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.3111905455589294, "clip_ratio/low_mean": 0.09228312410414219, "clip_ratio/low_min": 0.09228312410414219, "clip_ratio/high_mean": 0.11335889808833599, "clip_ratio/high_max": 0.11335889808833599, "clip_ratio/region_mean": 0.20564202219247818, "reward_total_mean": 0.13609255850315094, "reward_meter_mean": 0.3977913558483124, "reward_meter_std": 0.32337522506713867, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9964189529418945, "reward_repeat_soft_std": 0.0026952349580824375, "reward_judge_quality_mean": 0.22749999165534973, "reward_judge_quality_std": 0.02121320553123951, "reward_total_composite_mean": 0.13609255850315094, "reward_total_composite_std": 0.04253906011581421} {"timestamp_utc": "2026-04-13T13:01:48Z", "mode": "train", "global_step": 131, "epoch": 0.013159216474133602, "loss": 0.0339, "grad_norm": 24.785491943359375, "learning_rate": 9.606060606060607e-06, "num_tokens": 234316.0, "completions/mean_length": 28.875, "completions/min_length": 23.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.875, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.5526943206787109, "rewards/meter/std": 0.35744738578796387, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9920192360877991, "rewards/repeat_soft/std": 0.006725827232003212, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.13845448195934296, "rewards/total_composite/mean": 0.26369211077690125, "rewards/total_composite/std": 0.16711732745170593, "reward": 0.26369211077690125, "reward_std": 0.16711732745170593, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.26848104596138, "sampling/sampling_logp_difference/max": 4.247650146484375, "sampling/importance_sampling_ratio/min": 0.014297792688012123, "sampling/importance_sampling_ratio/mean": 0.9865650534629822, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6055498570203781, "clip_ratio/low_mean": 0.14761552587151527, "clip_ratio/low_min": 0.14761552587151527, "clip_ratio/high_mean": 0.0752564100548625, "clip_ratio/high_max": 0.0752564100548625, "clip_ratio/region_mean": 0.22287193592637777, "reward_total_mean": 0.26369211077690125, "reward_meter_mean": 0.5526943206787109, "reward_meter_std": 0.35744738578796387, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9920192360877991, "reward_repeat_soft_std": 0.006725827232003212, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.13845448195934296, "reward_total_composite_mean": 0.26369211077690125, "reward_total_composite_std": 0.16711732745170593} {"timestamp_utc": "2026-04-13T13:01:56Z", "mode": "train", "global_step": 132, "epoch": 0.013259668508287293, "loss": 0.0736, "grad_norm": 10.571706771850586, "learning_rate": 9.603030303030304e-06, "num_tokens": 236652.0, "completions/mean_length": 89.0, "completions/min_length": 76.0, "completions/max_length": 112.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.0, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 112.0, "rewards/meter/mean": 0.47271233797073364, "rewards/meter/std": 0.3446253836154938, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.994101881980896, "rewards/repeat_soft/std": 0.0062068188562989235, "rewards/judge_quality/mean": 0.23499999940395355, "rewards/judge_quality/std": 0.02777460403740406, "rewards/total_composite/mean": 0.15211760997772217, "rewards/total_composite/std": 0.04623790457844734, "reward": 0.15211760997772217, "reward_std": 0.04623790457844734, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21054711937904358, "sampling/sampling_logp_difference/max": 1.3416767120361328, "sampling/importance_sampling_ratio/min": 0.2614069879055023, "sampling/importance_sampling_ratio/mean": 1.0391924381256104, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.75347238779068, "clip_ratio/low_mean": 0.09103407897055149, "clip_ratio/low_min": 0.09103407897055149, "clip_ratio/high_mean": 0.10133074037730694, "clip_ratio/high_max": 0.10133074037730694, "clip_ratio/region_mean": 0.19236481934785843, "reward_total_mean": 0.15211760997772217, "reward_meter_mean": 0.47271233797073364, "reward_meter_std": 0.3446253836154938, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.994101881980896, "reward_repeat_soft_std": 0.0062068188562989235, "reward_judge_quality_mean": 0.23499999940395355, "reward_judge_quality_std": 0.02777460403740406, "reward_total_composite_mean": 0.15211760997772217, "reward_total_composite_std": 0.04623790457844734} {"timestamp_utc": "2026-04-13T13:02:03Z", "mode": "train", "global_step": 133, "epoch": 0.013360120542440984, "loss": 0.0575, "grad_norm": 15.153925895690918, "learning_rate": 9.600000000000001e-06, "num_tokens": 238734.0, "completions/mean_length": 61.25, "completions/min_length": 52.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.25, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.7094472646713257, "rewards/meter/std": 0.2717290222644806, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.99950110912323, "rewards/repeat_soft/std": 0.0009013467351906002, "rewards/judge_quality/mean": 0.23499999940395355, "rewards/judge_quality/std": 0.02777460403740406, "rewards/total_composite/mean": 0.19090116024017334, "rewards/total_composite/std": 0.048894003033638, "reward": 0.19090116024017334, "reward_std": 0.0488939993083477, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22949044406414032, "sampling/sampling_logp_difference/max": 1.3840246200561523, "sampling/importance_sampling_ratio/min": 0.25056806206703186, "sampling/importance_sampling_ratio/mean": 1.056289792060852, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.7774215042591095, "clip_ratio/low_mean": 0.08348813280463219, "clip_ratio/low_min": 0.08348813280463219, "clip_ratio/high_mean": 0.11317358072847128, "clip_ratio/high_max": 0.11317358072847128, "clip_ratio/region_mean": 0.19666171353310347, "reward_total_mean": 0.19090116024017334, "reward_meter_mean": 0.7094472646713257, "reward_meter_std": 0.2717290222644806, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.99950110912323, "reward_repeat_soft_std": 0.0009013467351906002, "reward_judge_quality_mean": 0.23499999940395355, "reward_judge_quality_std": 0.02777460403740406, "reward_total_composite_mean": 0.19090116024017334, "reward_total_composite_std": 0.048894003033638} {"timestamp_utc": "2026-04-13T13:02:10Z", "mode": "train", "global_step": 134, "epoch": 0.013460572576594675, "loss": -0.0098, "grad_norm": 9.651640892028809, "learning_rate": 9.596969696969699e-06, "num_tokens": 241075.0, "completions/mean_length": 99.625, "completions/min_length": 92.0, "completions/max_length": 112.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 99.625, "completions/min_terminated_length": 92.0, "completions/max_terminated_length": 112.0, "rewards/meter/mean": 0.9560881853103638, "rewards/meter/std": 0.0939415693283081, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9977115988731384, "rewards/repeat_soft/std": 0.0038488158024847507, "rewards/judge_quality/mean": 0.2199999988079071, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.18523168563842773, "rewards/total_composite/std": 0.0759832113981247, "reward": 0.18523168563842773, "reward_std": 0.0759832039475441, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23185819387435913, "sampling/sampling_logp_difference/max": 1.5198945999145508, "sampling/importance_sampling_ratio/min": 0.2187349498271942, "sampling/importance_sampling_ratio/mean": 1.0410938262939453, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.7694606482982635, "clip_ratio/low_mean": 0.04217284731566906, "clip_ratio/low_min": 0.04217284731566906, "clip_ratio/high_mean": 0.18183892592787743, "clip_ratio/high_max": 0.18183892592787743, "clip_ratio/region_mean": 0.22401177324354649, "reward_total_mean": 0.18523168563842773, "reward_meter_mean": 0.9560881853103638, "reward_meter_std": 0.0939415693283081, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9977115988731384, "reward_repeat_soft_std": 0.0038488158024847507, "reward_judge_quality_mean": 0.2199999988079071, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.18523168563842773, "reward_total_composite_std": 0.0759832113981247} {"timestamp_utc": "2026-04-13T13:02:17Z", "mode": "train", "global_step": 135, "epoch": 0.013561024610748368, "loss": 0.0927, "grad_norm": 10.741291999816895, "learning_rate": 9.593939393939394e-06, "num_tokens": 243080.0, "completions/mean_length": 84.625, "completions/min_length": 69.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.625, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.5973300933837891, "rewards/meter/std": 0.3906180262565613, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9978720545768738, "rewards/repeat_soft/std": 0.0020282238256186247, "rewards/judge_quality/mean": 0.20000000298023224, "rewards/judge_quality/std": 0.021380895748734474, "rewards/total_composite/mean": 0.14806291460990906, "rewards/total_composite/std": 0.05354469642043114, "reward": 0.14806291460990906, "reward_std": 0.05354469269514084, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23586569726467133, "sampling/sampling_logp_difference/max": 1.169741153717041, "sampling/importance_sampling_ratio/min": 0.3260723352432251, "sampling/importance_sampling_ratio/mean": 1.0564217567443848, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.5219001173973083, "clip_ratio/low_mean": 0.08202693238854408, "clip_ratio/low_min": 0.08202693238854408, "clip_ratio/high_mean": 0.11562965251505375, "clip_ratio/high_max": 0.11562965251505375, "clip_ratio/region_mean": 0.19765658490359783, "reward_total_mean": 0.14806291460990906, "reward_meter_mean": 0.5973300933837891, "reward_meter_std": 0.3906180262565613, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9978720545768738, "reward_repeat_soft_std": 0.0020282238256186247, "reward_judge_quality_mean": 0.20000000298023224, "reward_judge_quality_std": 0.021380895748734474, "reward_total_composite_mean": 0.14806291460990906, "reward_total_composite_std": 0.05354469642043114} {"timestamp_utc": "2026-04-13T13:02:24Z", "mode": "train", "global_step": 136, "epoch": 0.01366147664490206, "loss": 0.0637, "grad_norm": 13.939590454101562, "learning_rate": 9.590909090909091e-06, "num_tokens": 244994.0, "completions/mean_length": 59.25, "completions/min_length": 46.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.25, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.4887017607688904, "rewards/meter/std": 0.396069198846817, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9978152513504028, "rewards/repeat_soft/std": 0.003188708098605275, "rewards/judge_quality/mean": 0.41874998807907104, "rewards/judge_quality/std": 0.21596543490886688, "rewards/total_composite/mean": 0.27008306980133057, "rewards/total_composite/std": 0.1957637369632721, "reward": 0.27008306980133057, "reward_std": 0.1957637369632721, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21498402953147888, "sampling/sampling_logp_difference/max": 1.6607599258422852, "sampling/importance_sampling_ratio/min": 0.18999452888965607, "sampling/importance_sampling_ratio/mean": 1.046618103981018, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.5923212319612503, "clip_ratio/low_mean": 0.11506733112037182, "clip_ratio/low_min": 0.11506733112037182, "clip_ratio/high_mean": 0.11947667319327593, "clip_ratio/high_max": 0.11947667319327593, "clip_ratio/region_mean": 0.23454400431364775, "reward_total_mean": 0.27008306980133057, "reward_meter_mean": 0.4887017607688904, "reward_meter_std": 0.396069198846817, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9978152513504028, "reward_repeat_soft_std": 0.003188708098605275, "reward_judge_quality_mean": 0.41874998807907104, "reward_judge_quality_std": 0.21596543490886688, "reward_total_composite_mean": 0.27008306980133057, "reward_total_composite_std": 0.1957637369632721} {"timestamp_utc": "2026-04-13T13:02:31Z", "mode": "train", "global_step": 137, "epoch": 0.01376192867905575, "loss": 0.0725, "grad_norm": 20.54743194580078, "learning_rate": 9.587878787878789e-06, "num_tokens": 246492.0, "completions/mean_length": 33.25, "completions/min_length": 29.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.25, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.38131028413772583, "rewards/meter/std": 0.38800758123397827, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9944081902503967, "rewards/repeat_soft/std": 0.00883419718593359, "rewards/judge_quality/mean": 0.2524999976158142, "rewards/judge_quality/std": 0.07086203992366791, "rewards/total_composite/mean": 0.13826440274715424, "rewards/total_composite/std": 0.049926865845918655, "reward": 0.13826440274715424, "reward_std": 0.049926865845918655, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2485567033290863, "sampling/sampling_logp_difference/max": 1.640737771987915, "sampling/importance_sampling_ratio/min": 0.1938369870185852, "sampling/importance_sampling_ratio/mean": 1.0484215021133423, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.452331081032753, "clip_ratio/low_mean": 0.1011201636865735, "clip_ratio/low_min": 0.1011201636865735, "clip_ratio/high_mean": 0.05779710691422224, "clip_ratio/high_max": 0.05779710691422224, "clip_ratio/region_mean": 0.15891727060079575, "reward_total_mean": 0.13826440274715424, "reward_meter_mean": 0.38131028413772583, "reward_meter_std": 0.38800758123397827, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9944081902503967, "reward_repeat_soft_std": 0.00883419718593359, "reward_judge_quality_mean": 0.2524999976158142, "reward_judge_quality_std": 0.07086203992366791, "reward_total_composite_mean": 0.13826440274715424, "reward_total_composite_std": 0.049926865845918655} {"timestamp_utc": "2026-04-13T13:02:37Z", "mode": "train", "global_step": 138, "epoch": 0.013862380713209442, "loss": 0.0486, "grad_norm": 25.14579963684082, "learning_rate": 9.584848484848486e-06, "num_tokens": 248096.0, "completions/mean_length": 31.5, "completions/min_length": 27.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.5, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.45216143131256104, "rewards/meter/std": 0.3622969090938568, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9938479661941528, "rewards/repeat_soft/std": 0.008477476425468922, "rewards/judge_quality/mean": 0.26750001311302185, "rewards/judge_quality/std": 0.07497618347406387, "rewards/total_composite/mean": 0.17517951130867004, "rewards/total_composite/std": 0.0906360074877739, "reward": 0.17517951130867004, "reward_std": 0.0906360000371933, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22594934701919556, "sampling/sampling_logp_difference/max": 1.1099905967712402, "sampling/importance_sampling_ratio/min": 0.32956206798553467, "sampling/importance_sampling_ratio/mean": 1.0708420276641846, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.285018354654312, "clip_ratio/low_mean": 0.13230955973267555, "clip_ratio/low_min": 0.13230955973267555, "clip_ratio/high_mean": 0.09946177713572979, "clip_ratio/high_max": 0.09946177713572979, "clip_ratio/region_mean": 0.23177133686840534, "reward_total_mean": 0.17517951130867004, "reward_meter_mean": 0.45216143131256104, "reward_meter_std": 0.3622969090938568, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9938479661941528, "reward_repeat_soft_std": 0.008477476425468922, "reward_judge_quality_mean": 0.26750001311302185, "reward_judge_quality_std": 0.07497618347406387, "reward_total_composite_mean": 0.17517951130867004, "reward_total_composite_std": 0.0906360074877739} {"timestamp_utc": "2026-04-13T13:02:42Z", "mode": "train", "global_step": 139, "epoch": 0.013962832747363135, "loss": 0.0586, "grad_norm": 15.347241401672363, "learning_rate": 9.581818181818181e-06, "num_tokens": 249650.0, "completions/mean_length": 43.25, "completions/min_length": 35.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.25, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.512442946434021, "rewards/meter/std": 0.4792117178440094, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9996022582054138, "rewards/repeat_soft/std": 0.001124941511079669, "rewards/judge_quality/mean": 0.23499999940395355, "rewards/judge_quality/std": 0.01603567600250244, "rewards/total_composite/mean": 0.15505144000053406, "rewards/total_composite/std": 0.07194910943508148, "reward": 0.15505144000053406, "reward_std": 0.07194910943508148, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2576281726360321, "sampling/sampling_logp_difference/max": 1.5772953033447266, "sampling/importance_sampling_ratio/min": 0.20653294026851654, "sampling/importance_sampling_ratio/mean": 1.0444402694702148, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.1368479430675507, "clip_ratio/low_mean": 0.10965734161436558, "clip_ratio/low_min": 0.10965734161436558, "clip_ratio/high_mean": 0.1077451454475522, "clip_ratio/high_max": 0.1077451454475522, "clip_ratio/region_mean": 0.21740248706191778, "reward_total_mean": 0.15505144000053406, "reward_meter_mean": 0.512442946434021, "reward_meter_std": 0.4792117178440094, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9996022582054138, "reward_repeat_soft_std": 0.001124941511079669, "reward_judge_quality_mean": 0.23499999940395355, "reward_judge_quality_std": 0.01603567600250244, "reward_total_composite_mean": 0.15505144000053406, "reward_total_composite_std": 0.07194910943508148} {"timestamp_utc": "2026-04-13T13:02:49Z", "mode": "train", "global_step": 140, "epoch": 0.014063284781516826, "loss": 0.1773, "grad_norm": 31.55306053161621, "learning_rate": 9.57878787878788e-06, "num_tokens": 251181.0, "completions/mean_length": 33.375, "completions/min_length": 25.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.375, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.5141806602478027, "rewards/meter/std": 0.3833635151386261, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9908744096755981, "rewards/repeat_soft/std": 0.01445451844483614, "rewards/judge_quality/mean": 0.32749998569488525, "rewards/judge_quality/std": 0.21022097766399384, "rewards/total_composite/mean": 0.2127482146024704, "rewards/total_composite/std": 0.16522666811943054, "reward": 0.2127482146024704, "reward_std": 0.16522665321826935, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.29307568073272705, "sampling/sampling_logp_difference/max": 3.3791017532348633, "sampling/importance_sampling_ratio/min": 0.034078050404787064, "sampling/importance_sampling_ratio/mean": 0.9967068433761597, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4846941903233528, "clip_ratio/low_mean": 0.15150748565793037, "clip_ratio/low_min": 0.15150748565793037, "clip_ratio/high_mean": 0.06666666641831398, "clip_ratio/high_max": 0.06666666641831398, "clip_ratio/region_mean": 0.21817415207624435, "reward_total_mean": 0.2127482146024704, "reward_meter_mean": 0.5141806602478027, "reward_meter_std": 0.3833635151386261, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9908744096755981, "reward_repeat_soft_std": 0.01445451844483614, "reward_judge_quality_mean": 0.32749998569488525, "reward_judge_quality_std": 0.21022097766399384, "reward_total_composite_mean": 0.2127482146024704, "reward_total_composite_std": 0.16522666811943054} {"timestamp_utc": "2026-04-13T13:02:58Z", "mode": "train", "global_step": 141, "epoch": 0.014163736815670517, "loss": 0.0703, "grad_norm": 10.269293785095215, "learning_rate": 9.575757575757576e-06, "num_tokens": 253880.0, "completions/mean_length": 131.375, "completions/min_length": 114.0, "completions/max_length": 154.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.375, "completions/min_terminated_length": 114.0, "completions/max_terminated_length": 154.0, "rewards/meter/mean": 0.41031938791275024, "rewards/meter/std": 0.23784169554710388, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9922277927398682, "rewards/repeat_soft/std": 0.007713355589658022, "rewards/judge_quality/mean": 0.20749999582767487, "rewards/judge_quality/std": 0.0353553406894207, "rewards/total_composite/mean": 0.11861090362071991, "rewards/total_composite/std": 0.0585237592458725, "reward": 0.11861090362071991, "reward_std": 0.0585237555205822, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.239677295088768, "sampling/sampling_logp_difference/max": 1.5946722030639648, "sampling/importance_sampling_ratio/min": 0.2029750645160675, "sampling/importance_sampling_ratio/mean": 1.0555769205093384, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.2069272100925446, "clip_ratio/low_mean": 0.11101297475397587, "clip_ratio/low_min": 0.11101297475397587, "clip_ratio/high_mean": 0.12742743827402592, "clip_ratio/high_max": 0.12742743827402592, "clip_ratio/region_mean": 0.23844041302800179, "reward_total_mean": 0.11861090362071991, "reward_meter_mean": 0.41031938791275024, "reward_meter_std": 0.23784169554710388, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9922277927398682, "reward_repeat_soft_std": 0.007713355589658022, "reward_judge_quality_mean": 0.20749999582767487, "reward_judge_quality_std": 0.0353553406894207, "reward_total_composite_mean": 0.11861090362071991, "reward_total_composite_std": 0.0585237592458725} {"timestamp_utc": "2026-04-13T13:03:04Z", "mode": "train", "global_step": 142, "epoch": 0.014264188849824208, "loss": 0.1096, "grad_norm": 14.984911918640137, "learning_rate": 9.572727272727273e-06, "num_tokens": 255471.0, "completions/mean_length": 39.875, "completions/min_length": 35.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.875, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.8347629904747009, "rewards/meter/std": 0.2015080600976944, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9970236420631409, "rewards/repeat_soft/std": 0.005287198815494776, "rewards/judge_quality/mean": 0.2487500011920929, "rewards/judge_quality/std": 0.0699872374534607, "rewards/total_composite/mean": 0.20734521746635437, "rewards/total_composite/std": 0.10832992941141129, "reward": 0.20734521746635437, "reward_std": 0.10832992196083069, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23098348081111908, "sampling/sampling_logp_difference/max": 1.5200281143188477, "sampling/importance_sampling_ratio/min": 0.21870574355125427, "sampling/importance_sampling_ratio/mean": 1.059181571006775, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.3084492683410645, "clip_ratio/low_mean": 0.07645521871745586, "clip_ratio/low_min": 0.07645521871745586, "clip_ratio/high_mean": 0.12622801680117846, "clip_ratio/high_max": 0.12622801680117846, "clip_ratio/region_mean": 0.20268323551863432, "reward_total_mean": 0.20734521746635437, "reward_meter_mean": 0.8347629904747009, "reward_meter_std": 0.2015080600976944, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9970236420631409, "reward_repeat_soft_std": 0.005287198815494776, "reward_judge_quality_mean": 0.2487500011920929, "reward_judge_quality_std": 0.0699872374534607, "reward_total_composite_mean": 0.20734521746635437, "reward_total_composite_std": 0.10832992941141129} {"timestamp_utc": "2026-04-13T13:03:11Z", "mode": "train", "global_step": 143, "epoch": 0.014364640883977901, "loss": 0.0808, "grad_norm": 12.276087760925293, "learning_rate": 9.56969696969697e-06, "num_tokens": 257848.0, "completions/mean_length": 110.125, "completions/min_length": 96.0, "completions/max_length": 153.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 110.125, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 153.0, "rewards/meter/mean": 0.6634474992752075, "rewards/meter/std": 0.32899340987205505, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.996049165725708, "rewards/repeat_soft/std": 0.002863870235159993, "rewards/judge_quality/mean": 0.2224999964237213, "rewards/judge_quality/std": 0.027124052867293358, "rewards/total_composite/mean": 0.16733479499816895, "rewards/total_composite/std": 0.07748735696077347, "reward": 0.16733479499816895, "reward_std": 0.07748734951019287, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24100278317928314, "sampling/sampling_logp_difference/max": 1.3480043411254883, "sampling/importance_sampling_ratio/min": 0.2597581446170807, "sampling/importance_sampling_ratio/mean": 1.0553762912750244, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.7829255759716034, "clip_ratio/low_mean": 0.09594781510531902, "clip_ratio/low_min": 0.09594781510531902, "clip_ratio/high_mean": 0.1016995906829834, "clip_ratio/high_max": 0.1016995906829834, "clip_ratio/region_mean": 0.19764740578830242, "reward_total_mean": 0.16733479499816895, "reward_meter_mean": 0.6634474992752075, "reward_meter_std": 0.32899340987205505, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.996049165725708, "reward_repeat_soft_std": 0.002863870235159993, "reward_judge_quality_mean": 0.2224999964237213, "reward_judge_quality_std": 0.027124052867293358, "reward_total_composite_mean": 0.16733479499816895, "reward_total_composite_std": 0.07748735696077347} {"timestamp_utc": "2026-04-13T13:03:18Z", "mode": "train", "global_step": 144, "epoch": 0.014465092918131592, "loss": 0.0569, "grad_norm": 16.01355743408203, "learning_rate": 9.566666666666668e-06, "num_tokens": 259556.0, "completions/mean_length": 41.5, "completions/min_length": 34.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.5, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.34600621461868286, "rewards/meter/std": 0.3382630944252014, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9991676211357117, "rewards/repeat_soft/std": 0.0013059456832706928, "rewards/judge_quality/mean": 0.2724999785423279, "rewards/judge_quality/std": 0.0936177596449852, "rewards/total_composite/mean": 0.14613009989261627, "rewards/total_composite/std": 0.05449230223894119, "reward": 0.14613009989261627, "reward_std": 0.054492298513650894, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22496208548545837, "sampling/sampling_logp_difference/max": 1.772200584411621, "sampling/importance_sampling_ratio/min": 0.16995856165885925, "sampling/importance_sampling_ratio/mean": 1.0439956188201904, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.9176801443099976, "clip_ratio/low_mean": 0.08852885197848082, "clip_ratio/low_min": 0.08852885197848082, "clip_ratio/high_mean": 0.09302842989563942, "clip_ratio/high_max": 0.09302842989563942, "clip_ratio/region_mean": 0.18155728187412024, "reward_total_mean": 0.14613009989261627, "reward_meter_mean": 0.34600621461868286, "reward_meter_std": 0.3382630944252014, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9991676211357117, "reward_repeat_soft_std": 0.0013059456832706928, "reward_judge_quality_mean": 0.2724999785423279, "reward_judge_quality_std": 0.0936177596449852, "reward_total_composite_mean": 0.14613009989261627, "reward_total_composite_std": 0.05449230223894119} {"timestamp_utc": "2026-04-13T13:03:24Z", "mode": "train", "global_step": 145, "epoch": 0.014565544952285283, "loss": 0.0307, "grad_norm": 21.03781509399414, "learning_rate": 9.563636363636365e-06, "num_tokens": 261063.0, "completions/mean_length": 33.375, "completions/min_length": 28.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.375, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.26710832118988037, "rewards/meter/std": 0.30466634035110474, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9970107078552246, "rewards/repeat_soft/std": 0.003531090682372451, "rewards/judge_quality/mean": 0.32375001907348633, "rewards/judge_quality/std": 0.21206046640872955, "rewards/total_composite/mean": 0.16713756322860718, "rewards/total_composite/std": 0.12282278388738632, "reward": 0.16713756322860718, "reward_std": 0.12282278388738632, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23675867915153503, "sampling/sampling_logp_difference/max": 1.1583709716796875, "sampling/importance_sampling_ratio/min": 0.3139972984790802, "sampling/importance_sampling_ratio/mean": 0.9947248101234436, "sampling/importance_sampling_ratio/max": 1.9616643190383911, "entropy": 2.8966290950775146, "clip_ratio/low_mean": 0.11045025661587715, "clip_ratio/low_min": 0.11045025661587715, "clip_ratio/high_mean": 0.062420424073934555, "clip_ratio/high_max": 0.062420424073934555, "clip_ratio/region_mean": 0.1728706806898117, "reward_total_mean": 0.16713756322860718, "reward_meter_mean": 0.26710832118988037, "reward_meter_std": 0.30466634035110474, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9970107078552246, "reward_repeat_soft_std": 0.003531090682372451, "reward_judge_quality_mean": 0.32375001907348633, "reward_judge_quality_std": 0.21206046640872955, "reward_total_composite_mean": 0.16713756322860718, "reward_total_composite_std": 0.12282278388738632} {"timestamp_utc": "2026-04-13T13:03:30Z", "mode": "train", "global_step": 146, "epoch": 0.014665996986438976, "loss": -0.1402, "grad_norm": 15.96103286743164, "learning_rate": 9.56060606060606e-06, "num_tokens": 262641.0, "completions/mean_length": 28.25, "completions/min_length": 5.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 28.25, "completions/min_terminated_length": 5.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.21407319605350494, "rewards/meter/std": 0.17057731747627258, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9917581081390381, "rewards/repeat_soft/std": 0.01039243582636118, "rewards/judge_quality/mean": 0.26124998927116394, "rewards/judge_quality/std": 0.10091544687747955, "rewards/total_composite/mean": 0.12573854625225067, "rewards/total_composite/std": 0.07961715012788773, "reward": 0.12573854625225067, "reward_std": 0.07961715012788773, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2117980718612671, "sampling/sampling_logp_difference/max": 1.503438949584961, "sampling/importance_sampling_ratio/min": 0.222364142537117, "sampling/importance_sampling_ratio/mean": 1.0345395803451538, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.863342374563217, "clip_ratio/low_mean": 0.10940010473132133, "clip_ratio/low_min": 0.10940010473132133, "clip_ratio/high_mean": 0.06960083544254303, "clip_ratio/high_max": 0.06960083544254303, "clip_ratio/region_mean": 0.17900094017386436, "reward_total_mean": 0.12573854625225067, "reward_meter_mean": 0.21407319605350494, "reward_meter_std": 0.17057731747627258, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9917581081390381, "reward_repeat_soft_std": 0.01039243582636118, "reward_judge_quality_mean": 0.26124998927116394, "reward_judge_quality_std": 0.10091544687747955, "reward_total_composite_mean": 0.12573854625225067, "reward_total_composite_std": 0.07961715012788773} {"timestamp_utc": "2026-04-13T13:03:37Z", "mode": "train", "global_step": 147, "epoch": 0.014766449020592667, "loss": -0.0114, "grad_norm": 23.897436141967773, "learning_rate": 9.55757575757576e-06, "num_tokens": 264117.0, "completions/mean_length": 27.5, "completions/min_length": 23.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.5, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.6483590602874756, "rewards/meter/std": 0.38227975368499756, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9969677925109863, "rewards/repeat_soft/std": 0.005060871597379446, "rewards/judge_quality/mean": 0.2849999964237213, "rewards/judge_quality/std": 0.0843462198972702, "rewards/total_composite/mean": 0.20882880687713623, "rewards/total_composite/std": 0.11764669418334961, "reward": 0.20882880687713623, "reward_std": 0.11764669418334961, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.27704405784606934, "sampling/sampling_logp_difference/max": 2.9856948852539062, "sampling/importance_sampling_ratio/min": 0.050504397600889206, "sampling/importance_sampling_ratio/mean": 1.015399694442749, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8331507593393326, "clip_ratio/low_mean": 0.04557291604578495, "clip_ratio/low_min": 0.04557291604578495, "clip_ratio/high_mean": 0.1767373187467456, "clip_ratio/high_max": 0.1767373187467456, "clip_ratio/region_mean": 0.22231023479253054, "reward_total_mean": 0.20882880687713623, "reward_meter_mean": 0.6483590602874756, "reward_meter_std": 0.38227975368499756, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9969677925109863, "reward_repeat_soft_std": 0.005060871597379446, "reward_judge_quality_mean": 0.2849999964237213, "reward_judge_quality_std": 0.0843462198972702, "reward_total_composite_mean": 0.20882880687713623, "reward_total_composite_std": 0.11764669418334961} {"timestamp_utc": "2026-04-13T13:03:44Z", "mode": "train", "global_step": 148, "epoch": 0.014866901054746359, "loss": 0.033, "grad_norm": 10.57125186920166, "learning_rate": 9.554545454545455e-06, "num_tokens": 266314.0, "completions/mean_length": 85.625, "completions/min_length": 70.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.625, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.6430109739303589, "rewards/meter/std": 0.4566119611263275, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9956058263778687, "rewards/repeat_soft/std": 0.003198496997356415, "rewards/judge_quality/mean": 0.2449999898672104, "rewards/judge_quality/std": 0.0707106739282608, "rewards/total_composite/mean": 0.18349239230155945, "rewards/total_composite/std": 0.1234927698969841, "reward": 0.18349239230155945, "reward_std": 0.1234927624464035, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22205880284309387, "sampling/sampling_logp_difference/max": 1.428999900817871, "sampling/importance_sampling_ratio/min": 0.23954838514328003, "sampling/importance_sampling_ratio/mean": 1.0468692779541016, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.9796471297740936, "clip_ratio/low_mean": 0.06434272974729538, "clip_ratio/low_min": 0.06434272974729538, "clip_ratio/high_mean": 0.1586184985935688, "clip_ratio/high_max": 0.1586184985935688, "clip_ratio/region_mean": 0.22296122834086418, "reward_total_mean": 0.18349239230155945, "reward_meter_mean": 0.6430109739303589, "reward_meter_std": 0.4566119611263275, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9956058263778687, "reward_repeat_soft_std": 0.003198496997356415, "reward_judge_quality_mean": 0.2449999898672104, "reward_judge_quality_std": 0.0707106739282608, "reward_total_composite_mean": 0.18349239230155945, "reward_total_composite_std": 0.1234927698969841} {"timestamp_utc": "2026-04-13T13:03:51Z", "mode": "train", "global_step": 149, "epoch": 0.01496735308890005, "loss": -0.0165, "grad_norm": 8.9459228515625, "learning_rate": 9.551515151515152e-06, "num_tokens": 268823.0, "completions/mean_length": 122.625, "completions/min_length": 107.0, "completions/max_length": 160.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.625, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 160.0, "rewards/meter/mean": 0.8947402238845825, "rewards/meter/std": 0.14273923635482788, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.995945930480957, "rewards/repeat_soft/std": 0.004707190208137035, "rewards/judge_quality/mean": 0.21000000834465027, "rewards/judge_quality/std": 0.01851639896631241, "rewards/total_composite/mean": 0.19634748995304108, "rewards/total_composite/std": 0.03072374127805233, "reward": 0.19634748995304108, "reward_std": 0.03072374500334263, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23941372334957123, "sampling/sampling_logp_difference/max": 1.2456178665161133, "sampling/importance_sampling_ratio/min": 0.2877630591392517, "sampling/importance_sampling_ratio/mean": 1.0520843267440796, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.761257380247116, "clip_ratio/low_mean": 0.05636909045279026, "clip_ratio/low_min": 0.05636909045279026, "clip_ratio/high_mean": 0.16791298426687717, "clip_ratio/high_max": 0.16791298426687717, "clip_ratio/region_mean": 0.22428207471966743, "reward_total_mean": 0.19634748995304108, "reward_meter_mean": 0.8947402238845825, "reward_meter_std": 0.14273923635482788, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.995945930480957, "reward_repeat_soft_std": 0.004707190208137035, "reward_judge_quality_mean": 0.21000000834465027, "reward_judge_quality_std": 0.01851639896631241, "reward_total_composite_mean": 0.19634748995304108, "reward_total_composite_std": 0.03072374127805233} {"timestamp_utc": "2026-04-13T13:03:58Z", "mode": "train", "global_step": 150, "epoch": 0.015067805123053743, "loss": 0.0381, "grad_norm": 14.835107803344727, "learning_rate": 9.54848484848485e-06, "num_tokens": 270893.0, "completions/mean_length": 88.75, "completions/min_length": 69.0, "completions/max_length": 108.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 88.75, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 108.0, "rewards/meter/mean": 0.33144593238830566, "rewards/meter/std": 0.2663375735282898, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9982151389122009, "rewards/repeat_soft/std": 0.001292722998186946, "rewards/judge_quality/mean": 0.2799999713897705, "rewards/judge_quality/std": 0.09258200228214264, "rewards/total_composite/mean": 0.12507778406143188, "rewards/total_composite/std": 0.06719503551721573, "reward": 0.12507778406143188, "reward_std": 0.06719504296779633, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2821635901927948, "sampling/sampling_logp_difference/max": 1.908768653869629, "sampling/importance_sampling_ratio/min": 0.1482628434896469, "sampling/importance_sampling_ratio/mean": 1.0197124481201172, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.3200297951698303, "clip_ratio/low_mean": 0.1122272964566946, "clip_ratio/low_min": 0.1122272964566946, "clip_ratio/high_mean": 0.132456686347723, "clip_ratio/high_max": 0.132456686347723, "clip_ratio/region_mean": 0.2446839828044176, "reward_total_mean": 0.12507778406143188, "reward_meter_mean": 0.33144593238830566, "reward_meter_std": 0.2663375735282898, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9982151389122009, "reward_repeat_soft_std": 0.001292722998186946, "reward_judge_quality_mean": 0.2799999713897705, "reward_judge_quality_std": 0.09258200228214264, "reward_total_composite_mean": 0.12507778406143188, "reward_total_composite_std": 0.06719503551721573} {"timestamp_utc": "2026-04-13T13:04:39Z", "mode": "eval", "global_step": 150, "epoch": 0.015067805123053743, "eval_loss": NaN, "eval_runtime": 40.827, "eval_samples_per_second": 1.959, "eval_steps_per_second": 0.245, "eval_num_tokens": 270893.0, "eval_completions/mean_length": 65.85, "eval_completions/min_length": 25.5, "eval_completions/max_length": 129.7, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 65.85, "eval_completions/min_terminated_length": 25.5, "eval_completions/max_terminated_length": 129.7, "eval_rewards/meter/mean": 0.572745019197464, "eval_rewards/meter/std": 0.3921780318021774, "eval_rewards/count_adherence/mean": 0.9729166686534881, "eval_rewards/count_adherence/std": 0.07660323455929756, "eval_rewards/hard_gate/mean": 0.925, "eval_rewards/hard_gate/std": 0.18771235942840575, "eval_rewards/repeat_soft/mean": 0.9933066666126251, "eval_rewards/repeat_soft/std": 0.011911871808115393, "eval_rewards/judge_quality/mean": 0.26975000351667405, "eval_rewards/judge_quality/std": 0.09802335640415549, "eval_rewards/total_composite/mean": 0.1846210852265358, "eval_rewards/total_composite/std": 0.123112241178751, "eval_reward": 0.1846210852265358, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.17532754838466644, "eval_sampling/sampling_logp_difference/max": 1.2397335052490235, "eval_sampling/importance_sampling_ratio/min": 0.2943241521716118, "eval_sampling/importance_sampling_ratio/mean": 1.0524425029754638, "eval_sampling/importance_sampling_ratio/max": 1.6317439317703246, "eval_entropy": 3.0035820484161375, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.1846210852265358, "eval_reward_meter_mean": 0.572745019197464, "eval_reward_meter_std": 0.3921780318021774, "eval_reward_count_adherence_mean": 0.9729166686534881, "eval_reward_count_adherence_std": 0.07660323455929756, "eval_reward_hard_gate_mean": 0.925, "eval_reward_hard_gate_std": 0.18771235942840575, "eval_reward_repeat_soft_mean": 0.9933066666126251, "eval_reward_repeat_soft_std": 0.011911871808115393, "eval_reward_judge_quality_mean": 0.26975000351667405, "eval_reward_judge_quality_std": 0.09802335640415549, "eval_reward_total_composite_mean": 0.1846210852265358, "eval_reward_total_composite_std": 0.123112241178751} {"timestamp_utc": "2026-04-13T13:04:47Z", "mode": "train", "global_step": 151, "epoch": 0.015168257157207434, "loss": 0.0365, "grad_norm": 25.638601303100586, "learning_rate": 9.545454545454547e-06, "num_tokens": 272257.0, "completions/mean_length": 18.5, "completions/min_length": 14.0, "completions/max_length": 21.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 18.5, "completions/min_terminated_length": 14.0, "completions/max_terminated_length": 21.0, "rewards/meter/mean": 0.38172563910484314, "rewards/meter/std": 0.47442716360092163, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.2849999666213989, "rewards/judge_quality/std": 0.0843462198972702, "rewards/total_composite/mean": 0.17046958208084106, "rewards/total_composite/std": 0.10838013887405396, "reward": 0.17046958208084106, "reward_std": 0.10838013887405396, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23300570249557495, "sampling/sampling_logp_difference/max": 1.059906005859375, "sampling/importance_sampling_ratio/min": 0.3464883863925934, "sampling/importance_sampling_ratio/mean": 1.0775567293167114, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.6863486915826797, "clip_ratio/low_mean": 0.09908963739871979, "clip_ratio/low_min": 0.09908963739871979, "clip_ratio/high_mean": 0.10864662006497383, "clip_ratio/high_max": 0.10864662006497383, "clip_ratio/region_mean": 0.20773625746369362, "reward_total_mean": 0.17046958208084106, "reward_meter_mean": 0.38172563910484314, "reward_meter_std": 0.47442716360092163, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.2849999666213989, "reward_judge_quality_std": 0.0843462198972702, "reward_total_composite_mean": 0.17046958208084106, "reward_total_composite_std": 0.10838013887405396} {"timestamp_utc": "2026-04-13T13:04:54Z", "mode": "train", "global_step": 152, "epoch": 0.015268709191361125, "loss": 0.0344, "grad_norm": 8.511669158935547, "learning_rate": 9.542424242424242e-06, "num_tokens": 274275.0, "completions/mean_length": 89.25, "completions/min_length": 70.0, "completions/max_length": 123.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.25, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.8666473031044006, "rewards/meter/std": 0.2796079218387604, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9987285137176514, "rewards/repeat_soft/std": 0.002673062961548567, "rewards/judge_quality/mean": 0.2224999964237213, "rewards/judge_quality/std": 0.027124052867293358, "rewards/total_composite/mean": 0.17542394995689392, "rewards/total_composite/std": 0.08328276872634888, "reward": 0.17542394995689392, "reward_std": 0.08328276872634888, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22037450969219208, "sampling/sampling_logp_difference/max": 1.5642757415771484, "sampling/importance_sampling_ratio/min": 0.20923951268196106, "sampling/importance_sampling_ratio/mean": 1.0545034408569336, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.6310852766036987, "clip_ratio/low_mean": 0.04997062310576439, "clip_ratio/low_min": 0.04997062310576439, "clip_ratio/high_mean": 0.16189361177384853, "clip_ratio/high_max": 0.16189361177384853, "clip_ratio/region_mean": 0.21186423487961292, "reward_total_mean": 0.17542394995689392, "reward_meter_mean": 0.8666473031044006, "reward_meter_std": 0.2796079218387604, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9987285137176514, "reward_repeat_soft_std": 0.002673062961548567, "reward_judge_quality_mean": 0.2224999964237213, "reward_judge_quality_std": 0.027124052867293358, "reward_total_composite_mean": 0.17542394995689392, "reward_total_composite_std": 0.08328276872634888} {"timestamp_utc": "2026-04-13T13:05:00Z", "mode": "train", "global_step": 153, "epoch": 0.015369161225514816, "loss": 0.0429, "grad_norm": 15.867789268493652, "learning_rate": 9.539393939393941e-06, "num_tokens": 275767.0, "completions/mean_length": 40.5, "completions/min_length": 39.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.5, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.7182189226150513, "rewards/meter/std": 0.36515012383461, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9998095035552979, "rewards/repeat_soft/std": 0.000538805325049907, "rewards/judge_quality/mean": 0.2562499940395355, "rewards/judge_quality/std": 0.06781013309955597, "rewards/total_composite/mean": 0.19799037277698517, "rewards/total_composite/std": 0.040014926344156265, "reward": 0.19799037277698517, "reward_std": 0.040014926344156265, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22870279848575592, "sampling/sampling_logp_difference/max": 1.2410449981689453, "sampling/importance_sampling_ratio/min": 0.28908199071884155, "sampling/importance_sampling_ratio/mean": 1.0672976970672607, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.826114594936371, "clip_ratio/low_mean": 0.0574186984449625, "clip_ratio/low_min": 0.0574186984449625, "clip_ratio/high_mean": 0.1119742807932198, "clip_ratio/high_max": 0.1119742807932198, "clip_ratio/region_mean": 0.1693929792381823, "reward_total_mean": 0.19799037277698517, "reward_meter_mean": 0.7182189226150513, "reward_meter_std": 0.36515012383461, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9998095035552979, "reward_repeat_soft_std": 0.000538805325049907, "reward_judge_quality_mean": 0.2562499940395355, "reward_judge_quality_std": 0.06781013309955597, "reward_total_composite_mean": 0.19799037277698517, "reward_total_composite_std": 0.040014926344156265} {"timestamp_utc": "2026-04-13T13:05:06Z", "mode": "train", "global_step": 154, "epoch": 0.015469613259668509, "loss": 0.0105, "grad_norm": 16.742393493652344, "learning_rate": 9.536363636363637e-06, "num_tokens": 277471.0, "completions/mean_length": 41.0, "completions/min_length": 36.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.0, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.8258713483810425, "rewards/meter/std": 0.3087295591831207, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9977891445159912, "rewards/repeat_soft/std": 0.00410388084128499, "rewards/judge_quality/mean": 0.22374999523162842, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.19741278886795044, "rewards/total_composite/std": 0.04111725091934204, "reward": 0.19741278886795044, "reward_std": 0.04111725091934204, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21883884072303772, "sampling/sampling_logp_difference/max": 0.9221162796020508, "sampling/importance_sampling_ratio/min": 0.397676557302475, "sampling/importance_sampling_ratio/mean": 1.059522032737732, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.4892155528068542, "clip_ratio/low_mean": 0.03926282096654177, "clip_ratio/low_min": 0.03926282096654177, "clip_ratio/high_mean": 0.14135960396379232, "clip_ratio/high_max": 0.14135960396379232, "clip_ratio/region_mean": 0.1806224249303341, "reward_total_mean": 0.19741278886795044, "reward_meter_mean": 0.8258713483810425, "reward_meter_std": 0.3087295591831207, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9977891445159912, "reward_repeat_soft_std": 0.00410388084128499, "reward_judge_quality_mean": 0.22374999523162842, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.19741278886795044, "reward_total_composite_std": 0.04111725091934204} {"timestamp_utc": "2026-04-13T13:05:12Z", "mode": "train", "global_step": 155, "epoch": 0.0155700652938222, "loss": -0.0251, "grad_norm": 16.524307250976562, "learning_rate": 9.533333333333334e-06, "num_tokens": 279181.0, "completions/mean_length": 37.75, "completions/min_length": 33.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.75, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.6799507141113281, "rewards/meter/std": 0.3370908200740814, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9923340082168579, "rewards/repeat_soft/std": 0.007860827259719372, "rewards/judge_quality/mean": 0.30250000953674316, "rewards/judge_quality/std": 0.09808886796236038, "rewards/total_composite/mean": 0.24224551022052765, "rewards/total_composite/std": 0.11215207725763321, "reward": 0.24224551022052765, "reward_std": 0.1121520847082138, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23383033275604248, "sampling/sampling_logp_difference/max": 1.3716764450073242, "sampling/importance_sampling_ratio/min": 0.25368133187294006, "sampling/importance_sampling_ratio/mean": 1.077167272567749, "sampling/importance_sampling_ratio/max": 1.9759652614593506, "entropy": 3.1956107914447784, "clip_ratio/low_mean": 0.14641496166586876, "clip_ratio/low_min": 0.14641496166586876, "clip_ratio/high_mean": 0.08069717511534691, "clip_ratio/high_max": 0.08069717511534691, "clip_ratio/region_mean": 0.22711213678121567, "reward_total_mean": 0.24224551022052765, "reward_meter_mean": 0.6799507141113281, "reward_meter_std": 0.3370908200740814, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9923340082168579, "reward_repeat_soft_std": 0.007860827259719372, "reward_judge_quality_mean": 0.30250000953674316, "reward_judge_quality_std": 0.09808886796236038, "reward_total_composite_mean": 0.24224551022052765, "reward_total_composite_std": 0.11215207725763321} {"timestamp_utc": "2026-04-13T13:05:18Z", "mode": "train", "global_step": 156, "epoch": 0.01567051732797589, "loss": 0.0109, "grad_norm": 16.846054077148438, "learning_rate": 9.530303030303031e-06, "num_tokens": 280768.0, "completions/mean_length": 35.375, "completions/min_length": 33.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.375, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.47947975993156433, "rewards/meter/std": 0.40722402930259705, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9959161281585693, "rewards/repeat_soft/std": 0.006067224312573671, "rewards/judge_quality/mean": 0.29249998927116394, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.19201096892356873, "rewards/total_composite/std": 0.09286434203386307, "reward": 0.19201096892356873, "reward_std": 0.09286433458328247, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2250913381576538, "sampling/sampling_logp_difference/max": 2.343181848526001, "sampling/importance_sampling_ratio/min": 0.0960216298699379, "sampling/importance_sampling_ratio/mean": 1.0370150804519653, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.5058412104845047, "clip_ratio/low_mean": 0.1262683430686593, "clip_ratio/low_min": 0.1262683430686593, "clip_ratio/high_mean": 0.07929198257625103, "clip_ratio/high_max": 0.07929198257625103, "clip_ratio/region_mean": 0.20556032564491034, "reward_total_mean": 0.19201096892356873, "reward_meter_mean": 0.47947975993156433, "reward_meter_std": 0.40722402930259705, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9959161281585693, "reward_repeat_soft_std": 0.006067224312573671, "reward_judge_quality_mean": 0.29249998927116394, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.19201096892356873, "reward_total_composite_std": 0.09286434203386307} {"timestamp_utc": "2026-04-13T13:05:24Z", "mode": "train", "global_step": 157, "epoch": 0.015770969362129583, "loss": 0.1598, "grad_norm": 27.04548454284668, "learning_rate": 9.527272727272729e-06, "num_tokens": 282162.0, "completions/mean_length": 21.25, "completions/min_length": 17.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 21.25, "completions/min_terminated_length": 17.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.6387215852737427, "rewards/meter/std": 0.4759199023246765, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9669870734214783, "rewards/repeat_soft/std": 0.012691427022218704, "rewards/judge_quality/mean": 0.29249998927116394, "rewards/judge_quality/std": 0.08892211318016052, "rewards/total_composite/mean": 0.2150147259235382, "rewards/total_composite/std": 0.118369922041893, "reward": 0.2150147259235382, "reward_std": 0.11836991459131241, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2325924038887024, "sampling/sampling_logp_difference/max": 2.731811761856079, "sampling/importance_sampling_ratio/min": 0.06510123610496521, "sampling/importance_sampling_ratio/mean": 1.0563908815383911, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.7705140113830566, "clip_ratio/low_mean": 0.07694604247808456, "clip_ratio/low_min": 0.07694604247808456, "clip_ratio/high_mean": 0.14142475463449955, "clip_ratio/high_max": 0.14142475463449955, "clip_ratio/region_mean": 0.21837079711258411, "reward_total_mean": 0.2150147259235382, "reward_meter_mean": 0.6387215852737427, "reward_meter_std": 0.4759199023246765, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9669870734214783, "reward_repeat_soft_std": 0.012691427022218704, "reward_judge_quality_mean": 0.29249998927116394, "reward_judge_quality_std": 0.08892211318016052, "reward_total_composite_mean": 0.2150147259235382, "reward_total_composite_std": 0.118369922041893} {"timestamp_utc": "2026-04-13T13:05:30Z", "mode": "train", "global_step": 158, "epoch": 0.015871421396283274, "loss": 0.0504, "grad_norm": 15.0586519241333, "learning_rate": 9.524242424242424e-06, "num_tokens": 283658.0, "completions/mean_length": 39.0, "completions/min_length": 35.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.9479594230651855, "rewards/meter/std": 0.08010739833116531, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9995237588882446, "rewards/repeat_soft/std": 0.001279377960599959, "rewards/judge_quality/mean": 0.29499998688697815, "rewards/judge_quality/std": 0.1035098284482956, "rewards/total_composite/mean": 0.28706982731819153, "rewards/total_composite/std": 0.10837505757808685, "reward": 0.28706982731819153, "reward_std": 0.10837505757808685, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2167590707540512, "sampling/sampling_logp_difference/max": 1.1569538116455078, "sampling/importance_sampling_ratio/min": 0.31444257497787476, "sampling/importance_sampling_ratio/mean": 1.0546858310699463, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.4026435911655426, "clip_ratio/low_mean": 0.12356818560510874, "clip_ratio/low_min": 0.12356818560510874, "clip_ratio/high_mean": 0.07027027010917664, "clip_ratio/high_max": 0.07027027010917664, "clip_ratio/region_mean": 0.19383845571428537, "reward_total_mean": 0.28706982731819153, "reward_meter_mean": 0.9479594230651855, "reward_meter_std": 0.08010739833116531, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9995237588882446, "reward_repeat_soft_std": 0.001279377960599959, "reward_judge_quality_mean": 0.29499998688697815, "reward_judge_quality_std": 0.1035098284482956, "reward_total_composite_mean": 0.28706982731819153, "reward_total_composite_std": 0.10837505757808685} {"timestamp_utc": "2026-04-13T13:05:36Z", "mode": "train", "global_step": 159, "epoch": 0.015971873430436965, "loss": 0.0715, "grad_norm": 17.11335563659668, "learning_rate": 9.521212121212121e-06, "num_tokens": 285188.0, "completions/mean_length": 38.25, "completions/min_length": 35.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.25, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.9116860628128052, "rewards/meter/std": 0.20351704955101013, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9917818307876587, "rewards/repeat_soft/std": 0.017315642908215523, "rewards/judge_quality/mean": 0.30250000953674316, "rewards/judge_quality/std": 0.14508618414402008, "rewards/total_composite/mean": 0.27001631259918213, "rewards/total_composite/std": 0.08081734925508499, "reward": 0.27001631259918213, "reward_std": 0.08081734925508499, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21486471593379974, "sampling/sampling_logp_difference/max": 1.2924151420593262, "sampling/importance_sampling_ratio/min": 0.27460676431655884, "sampling/importance_sampling_ratio/mean": 1.033375859260559, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.1263866424560547, "clip_ratio/low_mean": 0.14018620643764734, "clip_ratio/low_min": 0.14018620643764734, "clip_ratio/high_mean": 0.05878378264605999, "clip_ratio/high_max": 0.05878378264605999, "clip_ratio/region_mean": 0.19896998908370733, "reward_total_mean": 0.27001631259918213, "reward_meter_mean": 0.9116860628128052, "reward_meter_std": 0.20351704955101013, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9917818307876587, "reward_repeat_soft_std": 0.017315642908215523, "reward_judge_quality_mean": 0.30250000953674316, "reward_judge_quality_std": 0.14508618414402008, "reward_total_composite_mean": 0.27001631259918213, "reward_total_composite_std": 0.08081734925508499} {"timestamp_utc": "2026-04-13T13:05:42Z", "mode": "train", "global_step": 160, "epoch": 0.01607232546459066, "loss": 0.1024, "grad_norm": 24.75937271118164, "learning_rate": 9.518181818181819e-06, "num_tokens": 286685.0, "completions/mean_length": 30.125, "completions/min_length": 26.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.125, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.6666136384010315, "rewards/meter/std": 0.42434483766555786, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9896146059036255, "rewards/repeat_soft/std": 0.011480763554573059, "rewards/judge_quality/mean": 0.45249998569488525, "rewards/judge_quality/std": 0.15554054081439972, "rewards/total_composite/mean": 0.36371999979019165, "rewards/total_composite/std": 0.18872661888599396, "reward": 0.36371999979019165, "reward_std": 0.18872661888599396, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18248780071735382, "sampling/sampling_logp_difference/max": 0.9513940811157227, "sampling/importance_sampling_ratio/min": 0.38620224595069885, "sampling/importance_sampling_ratio/mean": 1.059270977973938, "sampling/importance_sampling_ratio/max": 1.7763392925262451, "entropy": 2.211267575621605, "clip_ratio/low_mean": 0.07675521913915873, "clip_ratio/low_min": 0.07675521913915873, "clip_ratio/high_mean": 0.10903514828532934, "clip_ratio/high_max": 0.10903514828532934, "clip_ratio/region_mean": 0.18579036742448807, "reward_total_mean": 0.36371999979019165, "reward_meter_mean": 0.6666136384010315, "reward_meter_std": 0.42434483766555786, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9896146059036255, "reward_repeat_soft_std": 0.011480763554573059, "reward_judge_quality_mean": 0.45249998569488525, "reward_judge_quality_std": 0.15554054081439972, "reward_total_composite_mean": 0.36371999979019165, "reward_total_composite_std": 0.18872661888599396} {"timestamp_utc": "2026-04-13T13:05:48Z", "mode": "train", "global_step": 161, "epoch": 0.01617277749874435, "loss": 0.2212, "grad_norm": 24.690906524658203, "learning_rate": 9.515151515151516e-06, "num_tokens": 288121.0, "completions/mean_length": 33.5, "completions/min_length": 21.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.5, "completions/min_terminated_length": 21.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.5782473087310791, "rewards/meter/std": 0.40190985798835754, "rewards/count_adherence/mean": 0.5, "rewards/count_adherence/std": 0.5345224738121033, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9835919141769409, "rewards/repeat_soft/std": 0.018582111224532127, "rewards/judge_quality/mean": 0.25874999165534973, "rewards/judge_quality/std": 0.06978282332420349, "rewards/total_composite/mean": 0.15243111550807953, "rewards/total_composite/std": 0.1106199324131012, "reward": 0.15243111550807953, "reward_std": 0.1106199324131012, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.25054123997688293, "sampling/sampling_logp_difference/max": 1.4378957748413086, "sampling/importance_sampling_ratio/min": 0.23742683231830597, "sampling/importance_sampling_ratio/mean": 1.0478856563568115, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.502296507358551, "clip_ratio/low_mean": 0.09580637328326702, "clip_ratio/low_min": 0.09580637328326702, "clip_ratio/high_mean": 0.09887241385877132, "clip_ratio/high_max": 0.09887241385877132, "clip_ratio/region_mean": 0.19467878714203835, "reward_total_mean": 0.15243111550807953, "reward_meter_mean": 0.5782473087310791, "reward_meter_std": 0.40190985798835754, "reward_count_adherence_mean": 0.5, "reward_count_adherence_std": 0.5345224738121033, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9835919141769409, "reward_repeat_soft_std": 0.018582111224532127, "reward_judge_quality_mean": 0.25874999165534973, "reward_judge_quality_std": 0.06978282332420349, "reward_total_composite_mean": 0.15243111550807953, "reward_total_composite_std": 0.1106199324131012} {"timestamp_utc": "2026-04-13T13:05:54Z", "mode": "train", "global_step": 162, "epoch": 0.016273229532898042, "loss": 0.0518, "grad_norm": 22.90196418762207, "learning_rate": 9.512121212121213e-06, "num_tokens": 289618.0, "completions/mean_length": 27.125, "completions/min_length": 25.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.125, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.6658871173858643, "rewards/meter/std": 0.3142303228378296, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9950172305107117, "rewards/repeat_soft/std": 0.004112296737730503, "rewards/judge_quality/mean": 0.3100000023841858, "rewards/judge_quality/std": 0.2066052109003067, "rewards/total_composite/mean": 0.23733368515968323, "rewards/total_composite/std": 0.14951294660568237, "reward": 0.23733368515968323, "reward_std": 0.14951293170452118, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22878214716911316, "sampling/sampling_logp_difference/max": 1.463071346282959, "sampling/importance_sampling_ratio/min": 0.23152409493923187, "sampling/importance_sampling_ratio/mean": 1.0413239002227783, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9636720418930054, "clip_ratio/low_mean": 0.21846669167280197, "clip_ratio/low_min": 0.21846669167280197, "clip_ratio/high_mean": 0.03846153989434242, "clip_ratio/high_max": 0.03846153989434242, "clip_ratio/region_mean": 0.2569282315671444, "reward_total_mean": 0.23733368515968323, "reward_meter_mean": 0.6658871173858643, "reward_meter_std": 0.3142303228378296, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9950172305107117, "reward_repeat_soft_std": 0.004112296737730503, "reward_judge_quality_mean": 0.3100000023841858, "reward_judge_quality_std": 0.2066052109003067, "reward_total_composite_mean": 0.23733368515968323, "reward_total_composite_std": 0.14951294660568237} {"timestamp_utc": "2026-04-13T13:06:00Z", "mode": "train", "global_step": 163, "epoch": 0.016373681567051733, "loss": 0.0863, "grad_norm": 17.661081314086914, "learning_rate": 9.50909090909091e-06, "num_tokens": 291212.0, "completions/mean_length": 39.25, "completions/min_length": 33.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.25, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.810309648513794, "rewards/meter/std": 0.2990211844444275, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9958804845809937, "rewards/repeat_soft/std": 0.007363190874457359, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.17044061422348022, "rewards/total_composite/mean": 0.3469280004501343, "rewards/total_composite/std": 0.1944759637117386, "reward": 0.3469280004501343, "reward_std": 0.19447597861289978, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22683171927928925, "sampling/sampling_logp_difference/max": 1.625056505203247, "sampling/importance_sampling_ratio/min": 0.19690056145191193, "sampling/importance_sampling_ratio/mean": 1.0392217636108398, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.604361653327942, "clip_ratio/low_mean": 0.11104768421500921, "clip_ratio/low_min": 0.11104768421500921, "clip_ratio/high_mean": 0.10663313511759043, "clip_ratio/high_max": 0.10663313511759043, "clip_ratio/region_mean": 0.21768081933259964, "reward_total_mean": 0.3469280004501343, "reward_meter_mean": 0.810309648513794, "reward_meter_std": 0.2990211844444275, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9958804845809937, "reward_repeat_soft_std": 0.007363190874457359, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.17044061422348022, "reward_total_composite_mean": 0.3469280004501343, "reward_total_composite_std": 0.1944759637117386} {"timestamp_utc": "2026-04-13T13:06:06Z", "mode": "train", "global_step": 164, "epoch": 0.016474133601205424, "loss": -0.0344, "grad_norm": 10.669726371765137, "learning_rate": 9.506060606060606e-06, "num_tokens": 293394.0, "completions/mean_length": 85.75, "completions/min_length": 78.0, "completions/max_length": 107.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.75, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 107.0, "rewards/meter/mean": 0.7769610285758972, "rewards/meter/std": 0.32415905594825745, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.99647456407547, "rewards/repeat_soft/std": 0.0025675075594335794, "rewards/judge_quality/mean": 0.23499999940395355, "rewards/judge_quality/std": 0.02777460403740406, "rewards/total_composite/mean": 0.20259374380111694, "rewards/total_composite/std": 0.06132829189300537, "reward": 0.20259374380111694, "reward_std": 0.06132828816771507, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23315180838108063, "sampling/sampling_logp_difference/max": 1.7441082000732422, "sampling/importance_sampling_ratio/min": 0.1748007982969284, "sampling/importance_sampling_ratio/mean": 1.0587011575698853, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.6615799367427826, "clip_ratio/low_mean": 0.04825309570878744, "clip_ratio/low_min": 0.04825309570878744, "clip_ratio/high_mean": 0.12102040741592646, "clip_ratio/high_max": 0.12102040741592646, "clip_ratio/region_mean": 0.1692735031247139, "reward_total_mean": 0.20259374380111694, "reward_meter_mean": 0.7769610285758972, "reward_meter_std": 0.32415905594825745, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.99647456407547, "reward_repeat_soft_std": 0.0025675075594335794, "reward_judge_quality_mean": 0.23499999940395355, "reward_judge_quality_std": 0.02777460403740406, "reward_total_composite_mean": 0.20259374380111694, "reward_total_composite_std": 0.06132829189300537} {"timestamp_utc": "2026-04-13T13:06:12Z", "mode": "train", "global_step": 165, "epoch": 0.016574585635359115, "loss": -0.0044, "grad_norm": 23.314119338989258, "learning_rate": 9.503030303030303e-06, "num_tokens": 294906.0, "completions/mean_length": 31.0, "completions/min_length": 26.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.0, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.6155316233634949, "rewards/meter/std": 0.40069058537483215, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9974690079689026, "rewards/repeat_soft/std": 0.004686486907303333, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.13274572789669037, "rewards/total_composite/mean": 0.28638482093811035, "rewards/total_composite/std": 0.157676562666893, "reward": 0.28638482093811035, "reward_std": 0.157676562666893, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2523231506347656, "sampling/sampling_logp_difference/max": 2.2936909198760986, "sampling/importance_sampling_ratio/min": 0.10089338570833206, "sampling/importance_sampling_ratio/mean": 0.9666851162910461, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5693599730730057, "clip_ratio/low_mean": 0.11127274483442307, "clip_ratio/low_min": 0.11127274483442307, "clip_ratio/high_mean": 0.10227416455745697, "clip_ratio/high_max": 0.10227416455745697, "clip_ratio/region_mean": 0.21354690939188004, "reward_total_mean": 0.28638482093811035, "reward_meter_mean": 0.6155316233634949, "reward_meter_std": 0.40069058537483215, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9974690079689026, "reward_repeat_soft_std": 0.004686486907303333, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.13274572789669037, "reward_total_composite_mean": 0.28638482093811035, "reward_total_composite_std": 0.157676562666893} {"timestamp_utc": "2026-04-13T13:06:18Z", "mode": "train", "global_step": 166, "epoch": 0.016675037669512806, "loss": 0.1268, "grad_norm": 17.905162811279297, "learning_rate": 9.5e-06, "num_tokens": 296577.0, "completions/mean_length": 44.875, "completions/min_length": 34.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.875, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.33484289050102234, "rewards/meter/std": 0.37169912457466125, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9986193180084229, "rewards/repeat_soft/std": 0.002488938393071294, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.2686706781387329, "rewards/total_composite/mean": 0.22475406527519226, "rewards/total_composite/std": 0.22406691312789917, "reward": 0.22475406527519226, "reward_std": 0.22406691312789917, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23967714607715607, "sampling/sampling_logp_difference/max": 1.3330392837524414, "sampling/importance_sampling_ratio/min": 0.26367464661598206, "sampling/importance_sampling_ratio/mean": 1.035905122756958, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.881513684988022, "clip_ratio/low_mean": 0.13186770491302013, "clip_ratio/low_min": 0.13186770491302013, "clip_ratio/high_mean": 0.07607466168701649, "clip_ratio/high_max": 0.07607466168701649, "clip_ratio/region_mean": 0.20794236660003662, "reward_total_mean": 0.22475406527519226, "reward_meter_mean": 0.33484289050102234, "reward_meter_std": 0.37169912457466125, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9986193180084229, "reward_repeat_soft_std": 0.002488938393071294, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.2686706781387329, "reward_total_composite_mean": 0.22475406527519226, "reward_total_composite_std": 0.22406691312789917} {"timestamp_utc": "2026-04-13T13:06:25Z", "mode": "train", "global_step": 167, "epoch": 0.016775489703666498, "loss": 0.0488, "grad_norm": 13.672354698181152, "learning_rate": 9.496969696969698e-06, "num_tokens": 298093.0, "completions/mean_length": 42.5, "completions/min_length": 39.0, "completions/max_length": 57.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.5, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 57.0, "rewards/meter/mean": 0.9360833168029785, "rewards/meter/std": 0.06844120472669601, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9947550296783447, "rewards/repeat_soft/std": 0.010183374397456646, "rewards/judge_quality/mean": 0.28125, "rewards/judge_quality/std": 0.08675704896450043, "rewards/total_composite/mean": 0.26979902386665344, "rewards/total_composite/std": 0.08546984195709229, "reward": 0.26979902386665344, "reward_std": 0.08546984195709229, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23771153390407562, "sampling/sampling_logp_difference/max": 1.4594011306762695, "sampling/importance_sampling_ratio/min": 0.2323753982782364, "sampling/importance_sampling_ratio/mean": 1.0684665441513062, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.6879286766052246, "clip_ratio/low_mean": 0.1279240120202303, "clip_ratio/low_min": 0.1279240120202303, "clip_ratio/high_mean": 0.0570512842386961, "clip_ratio/high_max": 0.0570512842386961, "clip_ratio/region_mean": 0.1849752962589264, "reward_total_mean": 0.26979902386665344, "reward_meter_mean": 0.9360833168029785, "reward_meter_std": 0.06844120472669601, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9947550296783447, "reward_repeat_soft_std": 0.010183374397456646, "reward_judge_quality_mean": 0.28125, "reward_judge_quality_std": 0.08675704896450043, "reward_total_composite_mean": 0.26979902386665344, "reward_total_composite_std": 0.08546984195709229} {"timestamp_utc": "2026-04-13T13:06:31Z", "mode": "train", "global_step": 168, "epoch": 0.016875941737820192, "loss": -0.106, "grad_norm": 21.89627456665039, "learning_rate": 9.493939393939395e-06, "num_tokens": 299624.0, "completions/mean_length": 26.375, "completions/min_length": 19.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.375, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.647320032119751, "rewards/meter/std": 0.33076751232147217, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.4629100561141968, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9618749022483826, "rewards/repeat_soft/std": 0.024751795455813408, "rewards/judge_quality/mean": 0.2524999976158142, "rewards/judge_quality/std": 0.06902380287647247, "rewards/total_composite/mean": 0.17939558625221252, "rewards/total_composite/std": 0.06344479322433472, "reward": 0.17939558625221252, "reward_std": 0.06344479322433472, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2663666903972626, "sampling/sampling_logp_difference/max": 1.9853134155273438, "sampling/importance_sampling_ratio/min": 0.1373375654220581, "sampling/importance_sampling_ratio/mean": 1.0470249652862549, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.762563407421112, "clip_ratio/low_mean": 0.15201344713568687, "clip_ratio/low_min": 0.15201344713568687, "clip_ratio/high_mean": 0.0846162810921669, "clip_ratio/high_max": 0.0846162810921669, "clip_ratio/region_mean": 0.23662972822785378, "reward_total_mean": 0.17939558625221252, "reward_meter_mean": 0.647320032119751, "reward_meter_std": 0.33076751232147217, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.4629100561141968, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9618749022483826, "reward_repeat_soft_std": 0.024751795455813408, "reward_judge_quality_mean": 0.2524999976158142, "reward_judge_quality_std": 0.06902380287647247, "reward_total_composite_mean": 0.17939558625221252, "reward_total_composite_std": 0.06344479322433472} {"timestamp_utc": "2026-04-13T13:06:37Z", "mode": "train", "global_step": 169, "epoch": 0.016976393771973883, "loss": 0.2582, "grad_norm": 16.440635681152344, "learning_rate": 9.490909090909092e-06, "num_tokens": 301292.0, "completions/mean_length": 47.5, "completions/min_length": 34.0, "completions/max_length": 95.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.5, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 95.0, "rewards/meter/mean": 0.5857816934585571, "rewards/meter/std": 0.43221327662467957, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9991960525512695, "rewards/repeat_soft/std": 0.002273822668939829, "rewards/judge_quality/mean": 0.29874998331069946, "rewards/judge_quality/std": 0.10091544687747955, "rewards/total_composite/mean": 0.213999941945076, "rewards/total_composite/std": 0.09905601292848587, "reward": 0.213999941945076, "reward_std": 0.09905601292848587, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23288343846797943, "sampling/sampling_logp_difference/max": 0.9852371215820312, "sampling/importance_sampling_ratio/min": 0.37335067987442017, "sampling/importance_sampling_ratio/mean": 1.0595744848251343, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.978996843099594, "clip_ratio/low_mean": 0.04683104157447815, "clip_ratio/low_min": 0.04683104157447815, "clip_ratio/high_mean": 0.12344735581427813, "clip_ratio/high_max": 0.12344735581427813, "clip_ratio/region_mean": 0.17027839738875628, "reward_total_mean": 0.213999941945076, "reward_meter_mean": 0.5857816934585571, "reward_meter_std": 0.43221327662467957, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9991960525512695, "reward_repeat_soft_std": 0.002273822668939829, "reward_judge_quality_mean": 0.29874998331069946, "reward_judge_quality_std": 0.10091544687747955, "reward_total_composite_mean": 0.213999941945076, "reward_total_composite_std": 0.09905601292848587} {"timestamp_utc": "2026-04-13T13:06:43Z", "mode": "train", "global_step": 170, "epoch": 0.017076845806127575, "loss": 0.1449, "grad_norm": 14.68420124053955, "learning_rate": 9.487878787878788e-06, "num_tokens": 302793.0, "completions/mean_length": 40.625, "completions/min_length": 32.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.625, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.8819254636764526, "rewards/meter/std": 0.28798651695251465, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9958971738815308, "rewards/repeat_soft/std": 0.006846282631158829, "rewards/judge_quality/mean": 0.30250000953674316, "rewards/judge_quality/std": 0.14508618414402008, "rewards/total_composite/mean": 0.25590571761131287, "rewards/total_composite/std": 0.18720243871212006, "reward": 0.25590571761131287, "reward_std": 0.18720243871212006, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2317146360874176, "sampling/sampling_logp_difference/max": 1.2958250045776367, "sampling/importance_sampling_ratio/min": 0.2736719846725464, "sampling/importance_sampling_ratio/mean": 1.0399237871170044, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.074298083782196, "clip_ratio/low_mean": 0.15723508596420288, "clip_ratio/low_min": 0.15723508596420288, "clip_ratio/high_mean": 0.06250000186264515, "clip_ratio/high_max": 0.06250000186264515, "clip_ratio/region_mean": 0.21973508782684803, "reward_total_mean": 0.25590571761131287, "reward_meter_mean": 0.8819254636764526, "reward_meter_std": 0.28798651695251465, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9958971738815308, "reward_repeat_soft_std": 0.006846282631158829, "reward_judge_quality_mean": 0.30250000953674316, "reward_judge_quality_std": 0.14508618414402008, "reward_total_composite_mean": 0.25590571761131287, "reward_total_composite_std": 0.18720243871212006} {"timestamp_utc": "2026-04-13T13:06:49Z", "mode": "train", "global_step": 171, "epoch": 0.017177297840281266, "loss": 0.0779, "grad_norm": 16.1843318939209, "learning_rate": 9.484848484848485e-06, "num_tokens": 304440.0, "completions/mean_length": 39.875, "completions/min_length": 34.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.875, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.6370018124580383, "rewards/meter/std": 0.44355839490890503, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9982327222824097, "rewards/repeat_soft/std": 0.004276552703231573, "rewards/judge_quality/mean": 0.3062500059604645, "rewards/judge_quality/std": 0.2081165611743927, "rewards/total_composite/mean": 0.24946370720863342, "rewards/total_composite/std": 0.23714184761047363, "reward": 0.24946370720863342, "reward_std": 0.23714183270931244, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22012512385845184, "sampling/sampling_logp_difference/max": 1.327301025390625, "sampling/importance_sampling_ratio/min": 0.26519206166267395, "sampling/importance_sampling_ratio/mean": 1.0278286933898926, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.7738112658262253, "clip_ratio/low_mean": 0.15882117487490177, "clip_ratio/low_min": 0.15882117487490177, "clip_ratio/high_mean": 0.010135134682059288, "clip_ratio/high_max": 0.010135134682059288, "clip_ratio/region_mean": 0.16895630955696106, "reward_total_mean": 0.24946370720863342, "reward_meter_mean": 0.6370018124580383, "reward_meter_std": 0.44355839490890503, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9982327222824097, "reward_repeat_soft_std": 0.004276552703231573, "reward_judge_quality_mean": 0.3062500059604645, "reward_judge_quality_std": 0.2081165611743927, "reward_total_composite_mean": 0.24946370720863342, "reward_total_composite_std": 0.23714184761047363} {"timestamp_utc": "2026-04-13T13:06:55Z", "mode": "train", "global_step": 172, "epoch": 0.017277749874434957, "loss": 0.0646, "grad_norm": 16.315248489379883, "learning_rate": 9.481818181818182e-06, "num_tokens": 305980.0, "completions/mean_length": 29.5, "completions/min_length": 26.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.5, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.5681661367416382, "rewards/meter/std": 0.42752423882484436, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.964336097240448, "rewards/repeat_soft/std": 0.09348888695240021, "rewards/judge_quality/mean": 0.32374998927116394, "rewards/judge_quality/std": 0.1487027257680893, "rewards/total_composite/mean": 0.25733670592308044, "rewards/total_composite/std": 0.1935597062110901, "reward": 0.25733670592308044, "reward_std": 0.19355972111225128, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.26032719016075134, "sampling/sampling_logp_difference/max": 1.1938762664794922, "sampling/importance_sampling_ratio/min": 0.30304428935050964, "sampling/importance_sampling_ratio/mean": 1.0414456129074097, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.785316988825798, "clip_ratio/low_mean": 0.1700533702969551, "clip_ratio/low_min": 0.1700533702969551, "clip_ratio/high_mean": 0.09917582757771015, "clip_ratio/high_max": 0.09917582757771015, "clip_ratio/region_mean": 0.26922919787466526, "reward_total_mean": 0.25733670592308044, "reward_meter_mean": 0.5681661367416382, "reward_meter_std": 0.42752423882484436, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.964336097240448, "reward_repeat_soft_std": 0.09348888695240021, "reward_judge_quality_mean": 0.32374998927116394, "reward_judge_quality_std": 0.1487027257680893, "reward_total_composite_mean": 0.25733670592308044, "reward_total_composite_std": 0.1935597062110901} {"timestamp_utc": "2026-04-13T13:07:01Z", "mode": "train", "global_step": 173, "epoch": 0.017378201908588648, "loss": 0.0064, "grad_norm": 31.017641067504883, "learning_rate": 9.47878787878788e-06, "num_tokens": 307483.0, "completions/mean_length": 23.875, "completions/min_length": 19.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 23.875, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.712361216545105, "rewards/meter/std": 0.3987361788749695, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.95976322889328, "rewards/repeat_soft/std": 0.023058203980326653, "rewards/judge_quality/mean": 0.17250001430511475, "rewards/judge_quality/std": 0.02121320739388466, "rewards/total_composite/mean": 0.13786433637142181, "rewards/total_composite/std": 0.04557495191693306, "reward": 0.13786433637142181, "reward_std": 0.04557494819164276, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19148781895637512, "sampling/sampling_logp_difference/max": 2.123509407043457, "sampling/importance_sampling_ratio/min": 0.11961112171411514, "sampling/importance_sampling_ratio/mean": 1.0386604070663452, "sampling/importance_sampling_ratio/max": 1.9924687147140503, "entropy": 1.7448338121175766, "clip_ratio/low_mean": 0.061666667461395264, "clip_ratio/low_min": 0.061666667461395264, "clip_ratio/high_mean": 0.12374117225408554, "clip_ratio/high_max": 0.12374117225408554, "clip_ratio/region_mean": 0.1854078397154808, "reward_total_mean": 0.13786433637142181, "reward_meter_mean": 0.712361216545105, "reward_meter_std": 0.3987361788749695, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.95976322889328, "reward_repeat_soft_std": 0.023058203980326653, "reward_judge_quality_mean": 0.17250001430511475, "reward_judge_quality_std": 0.02121320739388466, "reward_total_composite_mean": 0.13786433637142181, "reward_total_composite_std": 0.04557495191693306} {"timestamp_utc": "2026-04-13T13:07:07Z", "mode": "train", "global_step": 174, "epoch": 0.01747865394274234, "loss": 0.0264, "grad_norm": 14.48010540008545, "learning_rate": 9.475757575757577e-06, "num_tokens": 309070.0, "completions/mean_length": 41.375, "completions/min_length": 35.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.375, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.5942786931991577, "rewards/meter/std": 0.27671706676483154, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9999185800552368, "rewards/repeat_soft/std": 0.0002302482316736132, "rewards/judge_quality/mean": 0.21875, "rewards/judge_quality/std": 0.018850916996598244, "rewards/total_composite/mean": 0.1614990532398224, "rewards/total_composite/std": 0.04365858435630798, "reward": 0.1614990532398224, "reward_std": 0.043658580631017685, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23007173836231232, "sampling/sampling_logp_difference/max": 1.1083860397338867, "sampling/importance_sampling_ratio/min": 0.33009129762649536, "sampling/importance_sampling_ratio/mean": 1.0569177865982056, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.840572714805603, "clip_ratio/low_mean": 0.07121522165834904, "clip_ratio/low_min": 0.07121522165834904, "clip_ratio/high_mean": 0.10277472622692585, "clip_ratio/high_max": 0.10277472622692585, "clip_ratio/region_mean": 0.1739899478852749, "reward_total_mean": 0.1614990532398224, "reward_meter_mean": 0.5942786931991577, "reward_meter_std": 0.27671706676483154, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9999185800552368, "reward_repeat_soft_std": 0.0002302482316736132, "reward_judge_quality_mean": 0.21875, "reward_judge_quality_std": 0.018850916996598244, "reward_total_composite_mean": 0.1614990532398224, "reward_total_composite_std": 0.04365858435630798} {"timestamp_utc": "2026-04-13T13:07:13Z", "mode": "train", "global_step": 175, "epoch": 0.01757910597689603, "loss": 0.0116, "grad_norm": 15.40805721282959, "learning_rate": 9.472727272727274e-06, "num_tokens": 310638.0, "completions/mean_length": 41.0, "completions/min_length": 30.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.0, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.7776926159858704, "rewards/meter/std": 0.36967718601226807, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9996999502182007, "rewards/repeat_soft/std": 0.0006752270855940878, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23754699528217316, "rewards/total_composite/mean": 0.4998127818107605, "rewards/total_composite/std": 0.2754271328449249, "reward": 0.4998127818107605, "reward_std": 0.2754271328449249, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1785004884004593, "sampling/sampling_logp_difference/max": 1.1226024627685547, "sampling/importance_sampling_ratio/min": 0.325431764125824, "sampling/importance_sampling_ratio/mean": 1.0309224128723145, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8760915398597717, "clip_ratio/low_mean": 0.0728583112359047, "clip_ratio/low_min": 0.0728583112359047, "clip_ratio/high_mean": 0.09241350181400776, "clip_ratio/high_max": 0.09241350181400776, "clip_ratio/region_mean": 0.16527181304991245, "reward_total_mean": 0.4998127818107605, "reward_meter_mean": 0.7776926159858704, "reward_meter_std": 0.36967718601226807, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9996999502182007, "reward_repeat_soft_std": 0.0006752270855940878, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23754699528217316, "reward_total_composite_mean": 0.4998127818107605, "reward_total_composite_std": 0.2754271328449249} {"timestamp_utc": "2026-04-13T13:07:19Z", "mode": "train", "global_step": 176, "epoch": 0.017679558011049725, "loss": 0.1406, "grad_norm": 25.277259826660156, "learning_rate": 9.469696969696971e-06, "num_tokens": 312172.0, "completions/mean_length": 33.75, "completions/min_length": 27.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.75, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.7242546081542969, "rewards/meter/std": 0.250786691904068, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9931771755218506, "rewards/repeat_soft/std": 0.013044352643191814, "rewards/judge_quality/mean": 0.3812499940395355, "rewards/judge_quality/std": 0.12799972295761108, "rewards/total_composite/mean": 0.30834031105041504, "rewards/total_composite/std": 0.11047318577766418, "reward": 0.30834031105041504, "reward_std": 0.11047318577766418, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24297761917114258, "sampling/sampling_logp_difference/max": 2.2212677001953125, "sampling/importance_sampling_ratio/min": 0.10847151279449463, "sampling/importance_sampling_ratio/mean": 1.0259636640548706, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.2932978942990303, "clip_ratio/low_mean": 0.1337986309081316, "clip_ratio/low_min": 0.1337986309081316, "clip_ratio/high_mean": 0.06702178157866001, "clip_ratio/high_max": 0.06702178157866001, "clip_ratio/region_mean": 0.2008204124867916, "reward_total_mean": 0.30834031105041504, "reward_meter_mean": 0.7242546081542969, "reward_meter_std": 0.250786691904068, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9931771755218506, "reward_repeat_soft_std": 0.013044352643191814, "reward_judge_quality_mean": 0.3812499940395355, "reward_judge_quality_std": 0.12799972295761108, "reward_total_composite_mean": 0.30834031105041504, "reward_total_composite_std": 0.11047318577766418} {"timestamp_utc": "2026-04-13T13:07:25Z", "mode": "train", "global_step": 177, "epoch": 0.017780010045203416, "loss": 0.1314, "grad_norm": 14.810586929321289, "learning_rate": 9.466666666666667e-06, "num_tokens": 313753.0, "completions/mean_length": 46.625, "completions/min_length": 39.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.625, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.45506995916366577, "rewards/meter/std": 0.43716365098953247, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9980074167251587, "rewards/repeat_soft/std": 0.005443721543997526, "rewards/judge_quality/mean": 0.2449999898672104, "rewards/judge_quality/std": 0.0707106739282608, "rewards/total_composite/mean": 0.16408401727676392, "rewards/total_composite/std": 0.11723174899816513, "reward": 0.16408401727676392, "reward_std": 0.11723174154758453, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24596381187438965, "sampling/sampling_logp_difference/max": 1.5142285823822021, "sampling/importance_sampling_ratio/min": 0.3284386396408081, "sampling/importance_sampling_ratio/mean": 1.0515755414962769, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.1561138927936554, "clip_ratio/low_mean": 0.11140831606462598, "clip_ratio/low_min": 0.11140831606462598, "clip_ratio/high_mean": 0.09245426766574383, "clip_ratio/high_max": 0.09245426766574383, "clip_ratio/region_mean": 0.2038625837303698, "reward_total_mean": 0.16408401727676392, "reward_meter_mean": 0.45506995916366577, "reward_meter_std": 0.43716365098953247, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9980074167251587, "reward_repeat_soft_std": 0.005443721543997526, "reward_judge_quality_mean": 0.2449999898672104, "reward_judge_quality_std": 0.0707106739282608, "reward_total_composite_mean": 0.16408401727676392, "reward_total_composite_std": 0.11723174899816513} {"timestamp_utc": "2026-04-13T13:07:33Z", "mode": "train", "global_step": 178, "epoch": 0.017880462079357107, "loss": 0.0457, "grad_norm": 8.597575187683105, "learning_rate": 9.463636363636364e-06, "num_tokens": 316168.0, "completions/mean_length": 103.875, "completions/min_length": 94.0, "completions/max_length": 155.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 103.875, "completions/min_terminated_length": 94.0, "completions/max_terminated_length": 155.0, "rewards/meter/mean": 0.8420461416244507, "rewards/meter/std": 0.28812357783317566, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.995198130607605, "rewards/repeat_soft/std": 0.0031533981673419476, "rewards/judge_quality/mean": 0.25999999046325684, "rewards/judge_quality/std": 0.07010196894407272, "rewards/total_composite/mean": 0.23466189205646515, "rewards/total_composite/std": 0.0839463621377945, "reward": 0.23466189205646515, "reward_std": 0.0839463621377945, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24356208741664886, "sampling/sampling_logp_difference/max": 1.4635021686553955, "sampling/importance_sampling_ratio/min": 0.23142436146736145, "sampling/importance_sampling_ratio/mean": 1.0635943412780762, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.91947278380394, "clip_ratio/low_mean": 0.12700920179486275, "clip_ratio/low_min": 0.12700920179486275, "clip_ratio/high_mean": 0.05360471457242966, "clip_ratio/high_max": 0.05360471457242966, "clip_ratio/region_mean": 0.1806139163672924, "reward_total_mean": 0.23466189205646515, "reward_meter_mean": 0.8420461416244507, "reward_meter_std": 0.28812357783317566, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.995198130607605, "reward_repeat_soft_std": 0.0031533981673419476, "reward_judge_quality_mean": 0.25999999046325684, "reward_judge_quality_std": 0.07010196894407272, "reward_total_composite_mean": 0.23466189205646515, "reward_total_composite_std": 0.0839463621377945} {"timestamp_utc": "2026-04-13T13:07:39Z", "mode": "train", "global_step": 179, "epoch": 0.0179809141135108, "loss": 0.0504, "grad_norm": 12.027117729187012, "learning_rate": 9.460606060606061e-06, "num_tokens": 318206.0, "completions/mean_length": 78.75, "completions/min_length": 73.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.75, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.5722213983535767, "rewards/meter/std": 0.2216475009918213, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9970940351486206, "rewards/repeat_soft/std": 0.00415117247030139, "rewards/judge_quality/mean": 0.2150000035762787, "rewards/judge_quality/std": 0.014142133295536041, "rewards/total_composite/mean": 0.12795740365982056, "rewards/total_composite/std": 0.05583568662405014, "reward": 0.12795740365982056, "reward_std": 0.05583568289875984, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22791191935539246, "sampling/sampling_logp_difference/max": 1.2603139877319336, "sampling/importance_sampling_ratio/min": 0.2835649847984314, "sampling/importance_sampling_ratio/mean": 1.0541539192199707, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.9370436668395996, "clip_ratio/low_mean": 0.044417669996619225, "clip_ratio/low_min": 0.044417669996619225, "clip_ratio/high_mean": 0.13506676629185677, "clip_ratio/high_max": 0.13506676629185677, "clip_ratio/region_mean": 0.179484436288476, "reward_total_mean": 0.12795740365982056, "reward_meter_mean": 0.5722213983535767, "reward_meter_std": 0.2216475009918213, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9970940351486206, "reward_repeat_soft_std": 0.00415117247030139, "reward_judge_quality_mean": 0.2150000035762787, "reward_judge_quality_std": 0.014142133295536041, "reward_total_composite_mean": 0.12795740365982056, "reward_total_composite_std": 0.05583568662405014} {"timestamp_utc": "2026-04-13T13:07:46Z", "mode": "train", "global_step": 180, "epoch": 0.01808136614766449, "loss": 0.0455, "grad_norm": 10.89418888092041, "learning_rate": 9.457575757575759e-06, "num_tokens": 320448.0, "completions/mean_length": 82.25, "completions/min_length": 73.0, "completions/max_length": 94.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.25, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.21472598612308502, "rewards/meter/std": 0.22956499457359314, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.12938730418682098, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.996832549571991, "rewards/repeat_soft/std": 0.001779663492925465, "rewards/judge_quality/mean": 0.2199999988079071, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.10347628593444824, "rewards/total_composite/std": 0.03403228893876076, "reward": 0.10347628593444824, "reward_std": 0.03403228521347046, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.25185060501098633, "sampling/sampling_logp_difference/max": 1.3912534713745117, "sampling/importance_sampling_ratio/min": 0.2487632930278778, "sampling/importance_sampling_ratio/mean": 1.0535268783569336, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.4406895339488983, "clip_ratio/low_mean": 0.14273057878017426, "clip_ratio/low_min": 0.14273057878017426, "clip_ratio/high_mean": 0.07947134785354137, "clip_ratio/high_max": 0.07947134785354137, "clip_ratio/region_mean": 0.22220192663371563, "reward_total_mean": 0.10347628593444824, "reward_meter_mean": 0.21472598612308502, "reward_meter_std": 0.22956499457359314, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.12938730418682098, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.996832549571991, "reward_repeat_soft_std": 0.001779663492925465, "reward_judge_quality_mean": 0.2199999988079071, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.10347628593444824, "reward_total_composite_std": 0.03403228893876076} {"timestamp_utc": "2026-04-13T13:07:52Z", "mode": "train", "global_step": 181, "epoch": 0.01818181818181818, "loss": 0.1036, "grad_norm": 16.209243774414062, "learning_rate": 9.454545454545456e-06, "num_tokens": 322063.0, "completions/mean_length": 41.875, "completions/min_length": 36.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.875, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.6623384952545166, "rewards/meter/std": 0.3999437093734741, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9959326982498169, "rewards/repeat_soft/std": 0.0059642670676112175, "rewards/judge_quality/mean": 0.3075000047683716, "rewards/judge_quality/std": 0.12464234232902527, "rewards/total_composite/mean": 0.25208374857902527, "rewards/total_composite/std": 0.14597024023532867, "reward": 0.25208374857902527, "reward_std": 0.14597022533416748, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2273944914340973, "sampling/sampling_logp_difference/max": 1.3746209144592285, "sampling/importance_sampling_ratio/min": 0.2529354691505432, "sampling/importance_sampling_ratio/mean": 1.0527184009552002, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.160714775323868, "clip_ratio/low_mean": 0.09416915010660887, "clip_ratio/low_min": 0.09416915010660887, "clip_ratio/high_mean": 0.08993902429938316, "clip_ratio/high_max": 0.08993902429938316, "clip_ratio/region_mean": 0.18410817440599203, "reward_total_mean": 0.25208374857902527, "reward_meter_mean": 0.6623384952545166, "reward_meter_std": 0.3999437093734741, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9959326982498169, "reward_repeat_soft_std": 0.0059642670676112175, "reward_judge_quality_mean": 0.3075000047683716, "reward_judge_quality_std": 0.12464234232902527, "reward_total_composite_mean": 0.25208374857902527, "reward_total_composite_std": 0.14597024023532867} {"timestamp_utc": "2026-04-13T13:07:58Z", "mode": "train", "global_step": 182, "epoch": 0.018282270215971872, "loss": 0.1077, "grad_norm": 23.082223892211914, "learning_rate": 9.451515151515153e-06, "num_tokens": 323623.0, "completions/mean_length": 31.0, "completions/min_length": 27.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.0, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.6646571159362793, "rewards/meter/std": 0.35609716176986694, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.998947024345398, "rewards/repeat_soft/std": 0.0015329680172726512, "rewards/judge_quality/mean": 0.2800000011920929, "rewards/judge_quality/std": 0.08976158499717712, "rewards/total_composite/mean": 0.2163580358028412, "rewards/total_composite/std": 0.11960655450820923, "reward": 0.2163580358028412, "reward_std": 0.11960654705762863, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24264128506183624, "sampling/sampling_logp_difference/max": 1.9742622375488281, "sampling/importance_sampling_ratio/min": 0.1388637274503708, "sampling/importance_sampling_ratio/mean": 1.0092780590057373, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5921870172023773, "clip_ratio/low_mean": 0.09682540036737919, "clip_ratio/low_min": 0.09682540036737919, "clip_ratio/high_mean": 0.10542328143492341, "clip_ratio/high_max": 0.10542328143492341, "clip_ratio/region_mean": 0.2022486818023026, "reward_total_mean": 0.2163580358028412, "reward_meter_mean": 0.6646571159362793, "reward_meter_std": 0.35609716176986694, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.998947024345398, "reward_repeat_soft_std": 0.0015329680172726512, "reward_judge_quality_mean": 0.2800000011920929, "reward_judge_quality_std": 0.08976158499717712, "reward_total_composite_mean": 0.2163580358028412, "reward_total_composite_std": 0.11960655450820923} {"timestamp_utc": "2026-04-13T13:08:04Z", "mode": "train", "global_step": 183, "epoch": 0.018382722250125567, "loss": 0.0646, "grad_norm": 23.999311447143555, "learning_rate": 9.448484848484849e-06, "num_tokens": 325260.0, "completions/mean_length": 31.625, "completions/min_length": 29.0, "completions/max_length": 36.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.625, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 36.0, "rewards/meter/mean": 0.7429101467132568, "rewards/meter/std": 0.30495375394821167, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9971073269844055, "rewards/repeat_soft/std": 0.005529557354748249, "rewards/judge_quality/mean": 0.3062499761581421, "rewards/judge_quality/std": 0.09500939399003983, "rewards/total_composite/mean": 0.24475175142288208, "rewards/total_composite/std": 0.07109037786722183, "reward": 0.24475175142288208, "reward_std": 0.07109037041664124, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17223596572875977, "sampling/sampling_logp_difference/max": 2.227447986602783, "sampling/importance_sampling_ratio/min": 0.10780319571495056, "sampling/importance_sampling_ratio/mean": 1.04634690284729, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.0528384000062943, "clip_ratio/low_mean": 0.11872574500739574, "clip_ratio/low_min": 0.11872574500739574, "clip_ratio/high_mean": 0.04085249127820134, "clip_ratio/high_max": 0.04085249127820134, "clip_ratio/region_mean": 0.15957823628559709, "reward_total_mean": 0.24475175142288208, "reward_meter_mean": 0.7429101467132568, "reward_meter_std": 0.30495375394821167, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9971073269844055, "reward_repeat_soft_std": 0.005529557354748249, "reward_judge_quality_mean": 0.3062499761581421, "reward_judge_quality_std": 0.09500939399003983, "reward_total_composite_mean": 0.24475175142288208, "reward_total_composite_std": 0.07109037786722183} {"timestamp_utc": "2026-04-13T13:08:10Z", "mode": "train", "global_step": 184, "epoch": 0.018483174284279258, "loss": 0.111, "grad_norm": 15.967321395874023, "learning_rate": 9.445454545454546e-06, "num_tokens": 326825.0, "completions/mean_length": 39.625, "completions/min_length": 31.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.625, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.5608224868774414, "rewards/meter/std": 0.357774555683136, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9991849064826965, "rewards/repeat_soft/std": 0.001608764985576272, "rewards/judge_quality/mean": 0.25999999046325684, "rewards/judge_quality/std": 0.06633248925209045, "rewards/total_composite/mean": 0.18913134932518005, "rewards/total_composite/std": 0.08285278081893921, "reward": 0.18913134932518005, "reward_std": 0.08285277336835861, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2188105583190918, "sampling/sampling_logp_difference/max": 1.127385139465332, "sampling/importance_sampling_ratio/min": 0.32387906312942505, "sampling/importance_sampling_ratio/mean": 1.0331828594207764, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.165031999349594, "clip_ratio/low_mean": 0.1111050397157669, "clip_ratio/low_min": 0.1111050397157669, "clip_ratio/high_mean": 0.10008255392313004, "clip_ratio/high_max": 0.10008255392313004, "clip_ratio/region_mean": 0.21118759363889694, "reward_total_mean": 0.18913134932518005, "reward_meter_mean": 0.5608224868774414, "reward_meter_std": 0.357774555683136, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9991849064826965, "reward_repeat_soft_std": 0.001608764985576272, "reward_judge_quality_mean": 0.25999999046325684, "reward_judge_quality_std": 0.06633248925209045, "reward_total_composite_mean": 0.18913134932518005, "reward_total_composite_std": 0.08285278081893921} {"timestamp_utc": "2026-04-13T13:08:15Z", "mode": "train", "global_step": 185, "epoch": 0.01858362631843295, "loss": 0.1385, "grad_norm": 20.29409408569336, "learning_rate": 9.442424242424243e-06, "num_tokens": 328275.0, "completions/mean_length": 29.25, "completions/min_length": 18.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.25, "completions/min_terminated_length": 18.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.5379096269607544, "rewards/meter/std": 0.43841618299484253, "rewards/count_adherence/mean": 0.5, "rewards/count_adherence/std": 0.5345224738121033, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9771194458007812, "rewards/repeat_soft/std": 0.01737549714744091, "rewards/judge_quality/mean": 0.2762500047683716, "rewards/judge_quality/std": 0.09179751574993134, "rewards/total_composite/mean": 0.17947977781295776, "rewards/total_composite/std": 0.1414765566587448, "reward": 0.17947977781295776, "reward_std": 0.1414765566587448, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2503550946712494, "sampling/sampling_logp_difference/max": 1.2705016136169434, "sampling/importance_sampling_ratio/min": 0.2806907892227173, "sampling/importance_sampling_ratio/mean": 1.0244457721710205, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.178879290819168, "clip_ratio/low_mean": 0.12165411468595266, "clip_ratio/low_min": 0.12165411468595266, "clip_ratio/high_mean": 0.11699196230620146, "clip_ratio/high_max": 0.11699196230620146, "clip_ratio/region_mean": 0.23864607699215412, "reward_total_mean": 0.17947977781295776, "reward_meter_mean": 0.5379096269607544, "reward_meter_std": 0.43841618299484253, "reward_count_adherence_mean": 0.5, "reward_count_adherence_std": 0.5345224738121033, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9771194458007812, "reward_repeat_soft_std": 0.01737549714744091, "reward_judge_quality_mean": 0.2762500047683716, "reward_judge_quality_std": 0.09179751574993134, "reward_total_composite_mean": 0.17947977781295776, "reward_total_composite_std": 0.1414765566587448} {"timestamp_utc": "2026-04-13T13:08:22Z", "mode": "train", "global_step": 186, "epoch": 0.01868407835258664, "loss": 0.0903, "grad_norm": 16.462566375732422, "learning_rate": 9.43939393939394e-06, "num_tokens": 330206.0, "completions/mean_length": 58.375, "completions/min_length": 49.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.375, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.27843326330184937, "rewards/meter/std": 0.245342418551445, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9970526695251465, "rewards/repeat_soft/std": 0.00237673451192677, "rewards/judge_quality/mean": 0.22624999284744263, "rewards/judge_quality/std": 0.02875388041138649, "rewards/total_composite/mean": 0.09679193049669266, "rewards/total_composite/std": 0.05075918138027191, "reward": 0.09679193049669266, "reward_std": 0.05075918138027191, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24853521585464478, "sampling/sampling_logp_difference/max": 2.083409309387207, "sampling/importance_sampling_ratio/min": 0.12450501322746277, "sampling/importance_sampling_ratio/mean": 1.0228904485702515, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.215415120124817, "clip_ratio/low_mean": 0.11344599910080433, "clip_ratio/low_min": 0.11344599910080433, "clip_ratio/high_mean": 0.11360197700560093, "clip_ratio/high_max": 0.11360197700560093, "clip_ratio/region_mean": 0.22704797610640526, "reward_total_mean": 0.09679193049669266, "reward_meter_mean": 0.27843326330184937, "reward_meter_std": 0.245342418551445, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9970526695251465, "reward_repeat_soft_std": 0.00237673451192677, "reward_judge_quality_mean": 0.22624999284744263, "reward_judge_quality_std": 0.02875388041138649, "reward_total_composite_mean": 0.09679193049669266, "reward_total_composite_std": 0.05075918138027191} {"timestamp_utc": "2026-04-13T13:08:29Z", "mode": "train", "global_step": 187, "epoch": 0.01878453038674033, "loss": 0.2065, "grad_norm": 6.505162239074707, "learning_rate": 9.436363636363636e-06, "num_tokens": 332657.0, "completions/mean_length": 108.375, "completions/min_length": 55.0, "completions/max_length": 142.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.375, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 142.0, "rewards/meter/mean": 0.5066986680030823, "rewards/meter/std": 0.2979062497615814, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9961463809013367, "rewards/repeat_soft/std": 0.0023854782339185476, "rewards/judge_quality/mean": 0.2449999898672104, "rewards/judge_quality/std": 0.0707106739282608, "rewards/total_composite/mean": 0.16829688847064972, "rewards/total_composite/std": 0.08634115755558014, "reward": 0.16829688847064972, "reward_std": 0.08634115010499954, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23701831698417664, "sampling/sampling_logp_difference/max": 1.195444107055664, "sampling/importance_sampling_ratio/min": 0.30256953835487366, "sampling/importance_sampling_ratio/mean": 1.079443335533142, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.9278726279735565, "clip_ratio/low_mean": 0.14010917581617832, "clip_ratio/low_min": 0.14010917581617832, "clip_ratio/high_mean": 0.05738133378326893, "clip_ratio/high_max": 0.05738133378326893, "clip_ratio/region_mean": 0.19749050959944725, "reward_total_mean": 0.16829688847064972, "reward_meter_mean": 0.5066986680030823, "reward_meter_std": 0.2979062497615814, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9961463809013367, "reward_repeat_soft_std": 0.0023854782339185476, "reward_judge_quality_mean": 0.2449999898672104, "reward_judge_quality_std": 0.0707106739282608, "reward_total_composite_mean": 0.16829688847064972, "reward_total_composite_std": 0.08634115755558014} {"timestamp_utc": "2026-04-13T13:08:35Z", "mode": "train", "global_step": 188, "epoch": 0.018884982420894023, "loss": 0.2508, "grad_norm": 15.58817195892334, "learning_rate": 9.433333333333335e-06, "num_tokens": 334275.0, "completions/mean_length": 45.25, "completions/min_length": 35.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.25, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.7768089771270752, "rewards/meter/std": 0.3373166024684906, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9966387748718262, "rewards/repeat_soft/std": 0.003940577618777752, "rewards/judge_quality/mean": 0.3199999928474426, "rewards/judge_quality/std": 0.10690449178218842, "rewards/total_composite/mean": 0.2800449728965759, "rewards/total_composite/std": 0.1464841514825821, "reward": 0.2800449728965759, "reward_std": 0.1464841365814209, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.218722403049469, "sampling/sampling_logp_difference/max": 1.2261724472045898, "sampling/importance_sampling_ratio/min": 0.2934134602546692, "sampling/importance_sampling_ratio/mean": 1.0738236904144287, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.576609343290329, "clip_ratio/low_mean": 0.10117842815816402, "clip_ratio/low_min": 0.10117842815816402, "clip_ratio/high_mean": 0.0984889529645443, "clip_ratio/high_max": 0.0984889529645443, "clip_ratio/region_mean": 0.19966738112270832, "reward_total_mean": 0.2800449728965759, "reward_meter_mean": 0.7768089771270752, "reward_meter_std": 0.3373166024684906, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9966387748718262, "reward_repeat_soft_std": 0.003940577618777752, "reward_judge_quality_mean": 0.3199999928474426, "reward_judge_quality_std": 0.10690449178218842, "reward_total_composite_mean": 0.2800449728965759, "reward_total_composite_std": 0.1464841514825821} {"timestamp_utc": "2026-04-13T13:08:41Z", "mode": "train", "global_step": 189, "epoch": 0.018985434455047714, "loss": 0.1027, "grad_norm": 25.442771911621094, "learning_rate": 9.43030303030303e-06, "num_tokens": 335768.0, "completions/mean_length": 30.625, "completions/min_length": 25.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.625, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.19121405482292175, "rewards/meter/std": 0.16347219049930573, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.993026852607727, "rewards/repeat_soft/std": 0.006410476751625538, "rewards/judge_quality/mean": 0.2199999988079071, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.10411349684000015, "rewards/total_composite/std": 0.02334478311240673, "reward": 0.10411349684000015, "reward_std": 0.02334478311240673, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.25707754492759705, "sampling/sampling_logp_difference/max": 1.7045478820800781, "sampling/importance_sampling_ratio/min": 0.18185459077358246, "sampling/importance_sampling_ratio/mean": 0.9856594204902649, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8737333565950394, "clip_ratio/low_mean": 0.10483114421367645, "clip_ratio/low_min": 0.10483114421367645, "clip_ratio/high_mean": 0.13011138327419758, "clip_ratio/high_max": 0.13011138327419758, "clip_ratio/region_mean": 0.23494252748787403, "reward_total_mean": 0.10411349684000015, "reward_meter_mean": 0.19121405482292175, "reward_meter_std": 0.16347219049930573, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.993026852607727, "reward_repeat_soft_std": 0.006410476751625538, "reward_judge_quality_mean": 0.2199999988079071, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.10411349684000015, "reward_total_composite_std": 0.02334478311240673} {"timestamp_utc": "2026-04-13T13:08:47Z", "mode": "train", "global_step": 190, "epoch": 0.019085886489201405, "loss": 0.0027, "grad_norm": 18.011791229248047, "learning_rate": 9.427272727272728e-06, "num_tokens": 337457.0, "completions/mean_length": 42.125, "completions/min_length": 31.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.125, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.44638946652412415, "rewards/meter/std": 0.3409777283668518, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9952579736709595, "rewards/repeat_soft/std": 0.008636967279016972, "rewards/judge_quality/mean": 0.23999999463558197, "rewards/judge_quality/std": 0.07406560331583023, "rewards/total_composite/mean": 0.14922666549682617, "rewards/total_composite/std": 0.056265976279973984, "reward": 0.14922666549682617, "reward_std": 0.056265972554683685, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23144479095935822, "sampling/sampling_logp_difference/max": 1.2533469200134277, "sampling/importance_sampling_ratio/min": 0.2855474650859833, "sampling/importance_sampling_ratio/mean": 1.0466233491897583, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.6538505852222443, "clip_ratio/low_mean": 0.10494689829647541, "clip_ratio/low_min": 0.10494689829647541, "clip_ratio/high_mean": 0.1201979462057352, "clip_ratio/high_max": 0.1201979462057352, "clip_ratio/region_mean": 0.22514484450221062, "reward_total_mean": 0.14922666549682617, "reward_meter_mean": 0.44638946652412415, "reward_meter_std": 0.3409777283668518, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9952579736709595, "reward_repeat_soft_std": 0.008636967279016972, "reward_judge_quality_mean": 0.23999999463558197, "reward_judge_quality_std": 0.07406560331583023, "reward_total_composite_mean": 0.14922666549682617, "reward_total_composite_std": 0.056265976279973984} {"timestamp_utc": "2026-04-13T13:08:55Z", "mode": "train", "global_step": 191, "epoch": 0.0191863385233551, "loss": 0.1239, "grad_norm": 9.168624877929688, "learning_rate": 9.424242424242425e-06, "num_tokens": 339804.0, "completions/mean_length": 108.375, "completions/min_length": 90.0, "completions/max_length": 131.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.375, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.6746603846549988, "rewards/meter/std": 0.3454626798629761, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9948081970214844, "rewards/repeat_soft/std": 0.004163301084190607, "rewards/judge_quality/mean": 0.2449999898672104, "rewards/judge_quality/std": 0.0707106739282608, "rewards/total_composite/mean": 0.17189136147499084, "rewards/total_composite/std": 0.08835437148809433, "reward": 0.17189136147499084, "reward_std": 0.08835437148809433, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24277395009994507, "sampling/sampling_logp_difference/max": 1.1603548526763916, "sampling/importance_sampling_ratio/min": 0.31337496638298035, "sampling/importance_sampling_ratio/mean": 1.060469627380371, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 4.501090317964554, "clip_ratio/low_mean": 0.07497364282608032, "clip_ratio/low_min": 0.07497364282608032, "clip_ratio/high_mean": 0.13471388444304466, "clip_ratio/high_max": 0.13471388444304466, "clip_ratio/region_mean": 0.20968752726912498, "reward_total_mean": 0.17189136147499084, "reward_meter_mean": 0.6746603846549988, "reward_meter_std": 0.3454626798629761, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9948081970214844, "reward_repeat_soft_std": 0.004163301084190607, "reward_judge_quality_mean": 0.2449999898672104, "reward_judge_quality_std": 0.0707106739282608, "reward_total_composite_mean": 0.17189136147499084, "reward_total_composite_std": 0.08835437148809433} {"timestamp_utc": "2026-04-13T13:09:01Z", "mode": "train", "global_step": 192, "epoch": 0.01928679055750879, "loss": 0.1073, "grad_norm": 26.564390182495117, "learning_rate": 9.421212121212122e-06, "num_tokens": 341388.0, "completions/mean_length": 27.0, "completions/min_length": 22.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.0, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.5459522604942322, "rewards/meter/std": 0.4349660873413086, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9664831757545471, "rewards/repeat_soft/std": 0.08932723104953766, "rewards/judge_quality/mean": 0.3700000047683716, "rewards/judge_quality/std": 0.1414213478565216, "rewards/total_composite/mean": 0.24714788794517517, "rewards/total_composite/std": 0.11071964353322983, "reward": 0.24714788794517517, "reward_std": 0.11071963608264923, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2507804334163666, "sampling/sampling_logp_difference/max": 2.2007923126220703, "sampling/importance_sampling_ratio/min": 0.11071540415287018, "sampling/importance_sampling_ratio/mean": 1.0464229583740234, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2436086535453796, "clip_ratio/low_mean": 0.10971423983573914, "clip_ratio/low_min": 0.10971423983573914, "clip_ratio/high_mean": 0.10494063422083855, "clip_ratio/high_max": 0.10494063422083855, "clip_ratio/region_mean": 0.21465487405657768, "reward_total_mean": 0.24714788794517517, "reward_meter_mean": 0.5459522604942322, "reward_meter_std": 0.4349660873413086, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9664831757545471, "reward_repeat_soft_std": 0.08932723104953766, "reward_judge_quality_mean": 0.3700000047683716, "reward_judge_quality_std": 0.1414213478565216, "reward_total_composite_mean": 0.24714788794517517, "reward_total_composite_std": 0.11071964353322983} {"timestamp_utc": "2026-04-13T13:09:09Z", "mode": "train", "global_step": 193, "epoch": 0.019387242591662482, "loss": 0.6028, "grad_norm": 12.519133567810059, "learning_rate": 9.418181818181818e-06, "num_tokens": 343477.0, "completions/mean_length": 83.125, "completions/min_length": 45.0, "completions/max_length": 292.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 83.125, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 292.0, "rewards/meter/mean": 0.6665387153625488, "rewards/meter/std": 0.38474684953689575, "rewards/count_adherence/mean": 0.7916666865348816, "rewards/count_adherence/std": 0.3535533845424652, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9970675110816956, "rewards/repeat_soft/std": 0.0038372064009308815, "rewards/judge_quality/mean": 0.23999999463558197, "rewards/judge_quality/std": 0.08485281467437744, "rewards/total_composite/mean": 0.19229841232299805, "rewards/total_composite/std": 0.11471903324127197, "reward": 0.19229841232299805, "reward_std": 0.11471902579069138, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.24479778110980988, "sampling/sampling_logp_difference/max": 1.2485408782958984, "sampling/importance_sampling_ratio/min": 0.2869231402873993, "sampling/importance_sampling_ratio/mean": 1.0734835863113403, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.8438830971717834, "clip_ratio/low_mean": 0.07805478479713202, "clip_ratio/low_min": 0.07805478479713202, "clip_ratio/high_mean": 0.10036757402122021, "clip_ratio/high_max": 0.10036757402122021, "clip_ratio/region_mean": 0.17842235881835222, "reward_total_mean": 0.19229841232299805, "reward_meter_mean": 0.6665387153625488, "reward_meter_std": 0.38474684953689575, "reward_count_adherence_mean": 0.7916666865348816, "reward_count_adherence_std": 0.3535533845424652, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9970675110816956, "reward_repeat_soft_std": 0.0038372064009308815, "reward_judge_quality_mean": 0.23999999463558197, "reward_judge_quality_std": 0.08485281467437744, "reward_total_composite_mean": 0.19229841232299805, "reward_total_composite_std": 0.11471903324127197} {"timestamp_utc": "2026-04-13T13:09:20Z", "mode": "train", "global_step": 194, "epoch": 0.019487694625816173, "loss": 0.3784, "grad_norm": 5.710408687591553, "learning_rate": 9.415151515151515e-06, "num_tokens": 345504.0, "completions/mean_length": 160.375, "completions/min_length": 35.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 110.14286041259766, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 407.0, "rewards/meter/mean": 0.3032819330692291, "rewards/meter/std": 0.31939348578453064, "rewards/count_adherence/mean": 0.6875, "rewards/count_adherence/std": 0.45806270837783813, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9990205764770508, "rewards/repeat_soft/std": 0.00227516982704401, "rewards/judge_quality/mean": 0.2524999976158142, "rewards/judge_quality/std": 0.14694508910179138, "rewards/total_composite/mean": 0.14139440655708313, "rewards/total_composite/std": 0.13582220673561096, "reward": 0.14139440655708313, "reward_std": 0.13582219183444977, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.23204858601093292, "sampling/sampling_logp_difference/max": 1.1267271041870117, "sampling/importance_sampling_ratio/min": 0.32409223914146423, "sampling/importance_sampling_ratio/mean": 1.0642539262771606, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.8016223311424255, "clip_ratio/low_mean": 0.03341209935024381, "clip_ratio/low_min": 0.03341209935024381, "clip_ratio/high_mean": 0.08239117078483105, "clip_ratio/high_max": 0.08239117078483105, "clip_ratio/region_mean": 0.11580327013507485, "reward_total_mean": 0.14139440655708313, "reward_meter_mean": 0.3032819330692291, "reward_meter_std": 0.31939348578453064, "reward_count_adherence_mean": 0.6875, "reward_count_adherence_std": 0.45806270837783813, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9990205764770508, "reward_repeat_soft_std": 0.00227516982704401, "reward_judge_quality_mean": 0.2524999976158142, "reward_judge_quality_std": 0.14694508910179138, "reward_total_composite_mean": 0.14139440655708313, "reward_total_composite_std": 0.13582220673561096} {"timestamp_utc": "2026-04-13T13:09:27Z", "mode": "train", "global_step": 195, "epoch": 0.019588146659969864, "loss": 0.1807, "grad_norm": 11.035167694091797, "learning_rate": 9.412121212121212e-06, "num_tokens": 347212.0, "completions/mean_length": 54.5, "completions/min_length": 38.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.5, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.21689274907112122, "rewards/meter/std": 0.2749010920524597, "rewards/count_adherence/mean": 0.6875, "rewards/count_adherence/std": 0.25877460837364197, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9974724054336548, "rewards/repeat_soft/std": 0.004348366055637598, "rewards/judge_quality/mean": 0.22749999165534973, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.09759880602359772, "rewards/total_composite/std": 0.047488220036029816, "reward": 0.09759880602359772, "reward_std": 0.04748821631073952, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.25424426794052124, "sampling/sampling_logp_difference/max": 1.341461181640625, "sampling/importance_sampling_ratio/min": 0.26146334409713745, "sampling/importance_sampling_ratio/mean": 1.063415288925171, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.7377142012119293, "clip_ratio/low_mean": 0.17252291180193424, "clip_ratio/low_min": 0.17252291180193424, "clip_ratio/high_mean": 0.06174089200794697, "clip_ratio/high_max": 0.06174089200794697, "clip_ratio/region_mean": 0.2342638038098812, "reward_total_mean": 0.09759880602359772, "reward_meter_mean": 0.21689274907112122, "reward_meter_std": 0.2749010920524597, "reward_count_adherence_mean": 0.6875, "reward_count_adherence_std": 0.25877460837364197, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9974724054336548, "reward_repeat_soft_std": 0.004348366055637598, "reward_judge_quality_mean": 0.22749999165534973, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.09759880602359772, "reward_total_composite_std": 0.047488220036029816} {"timestamp_utc": "2026-04-13T13:09:32Z", "mode": "train", "global_step": 196, "epoch": 0.019688598694123555, "loss": 0.0586, "grad_norm": 20.3045711517334, "learning_rate": 9.40909090909091e-06, "num_tokens": 348558.0, "completions/mean_length": 19.25, "completions/min_length": 15.0, "completions/max_length": 24.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 19.25, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 24.0, "rewards/meter/mean": 0.5030122995376587, "rewards/meter/std": 0.4006531834602356, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.47749999165534973, "rewards/judge_quality/std": 0.27027764916419983, "rewards/total_composite/mean": 0.3441108763217926, "rewards/total_composite/std": 0.25904008746147156, "reward": 0.3441108763217926, "reward_std": 0.25904008746147156, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18175098299980164, "sampling/sampling_logp_difference/max": 1.2526583671569824, "sampling/importance_sampling_ratio/min": 0.28574416041374207, "sampling/importance_sampling_ratio/mean": 1.0194385051727295, "sampling/importance_sampling_ratio/max": 1.8192806243896484, "entropy": 2.041719302535057, "clip_ratio/low_mean": 0.12170269526541233, "clip_ratio/low_min": 0.12170269526541233, "clip_ratio/high_mean": 0.029495615046471357, "clip_ratio/high_max": 0.029495615046471357, "clip_ratio/region_mean": 0.1511983103118837, "reward_total_mean": 0.3441108763217926, "reward_meter_mean": 0.5030122995376587, "reward_meter_std": 0.4006531834602356, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.47749999165534973, "reward_judge_quality_std": 0.27027764916419983, "reward_total_composite_mean": 0.3441108763217926, "reward_total_composite_std": 0.25904008746147156} {"timestamp_utc": "2026-04-13T13:09:38Z", "mode": "train", "global_step": 197, "epoch": 0.019789050728277247, "loss": 0.1945, "grad_norm": 13.562896728515625, "learning_rate": 9.406060606060607e-06, "num_tokens": 349978.0, "completions/mean_length": 30.5, "completions/min_length": 16.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.5, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.8292700052261353, "rewards/meter/std": 0.32876673340797424, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.5175492167472839, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9738503098487854, "rewards/repeat_soft/std": 0.021921547129750252, "rewards/judge_quality/mean": 0.45499998331069946, "rewards/judge_quality/std": 0.3196873664855957, "rewards/total_composite/mean": 0.4093332290649414, "rewards/total_composite/std": 0.3494712710380554, "reward": 0.4093332290649414, "reward_std": 0.34947124123573303, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22632241249084473, "sampling/sampling_logp_difference/max": 1.3101892471313477, "sampling/importance_sampling_ratio/min": 0.26976901292800903, "sampling/importance_sampling_ratio/mean": 1.0688990354537964, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.687138594686985, "clip_ratio/low_mean": 0.12821621261537075, "clip_ratio/low_min": 0.12821621261537075, "clip_ratio/high_mean": 0.06647727452218533, "clip_ratio/high_max": 0.06647727452218533, "clip_ratio/region_mean": 0.19469348713755608, "reward_total_mean": 0.4093332290649414, "reward_meter_mean": 0.8292700052261353, "reward_meter_std": 0.32876673340797424, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.5175492167472839, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9738503098487854, "reward_repeat_soft_std": 0.021921547129750252, "reward_judge_quality_mean": 0.45499998331069946, "reward_judge_quality_std": 0.3196873664855957, "reward_total_composite_mean": 0.4093332290649414, "reward_total_composite_std": 0.3494712710380554} {"timestamp_utc": "2026-04-13T13:09:44Z", "mode": "train", "global_step": 198, "epoch": 0.019889502762430938, "loss": 0.0457, "grad_norm": 18.57405662536621, "learning_rate": 9.403030303030304e-06, "num_tokens": 351830.0, "completions/mean_length": 48.5, "completions/min_length": 40.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.5, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.3581885099411011, "rewards/meter/std": 0.3801507353782654, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.993811845779419, "rewards/repeat_soft/std": 0.005281467922031879, "rewards/judge_quality/mean": 0.3799999952316284, "rewards/judge_quality/std": 0.2112547606229782, "rewards/total_composite/mean": 0.2376668006181717, "rewards/total_composite/std": 0.22460903227329254, "reward": 0.2376668006181717, "reward_std": 0.22460903227329254, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.214599147439003, "sampling/sampling_logp_difference/max": 3.266845226287842, "sampling/importance_sampling_ratio/min": 0.038126517087221146, "sampling/importance_sampling_ratio/mean": 0.9915326833724976, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.411556027829647, "clip_ratio/low_mean": 0.1614802423864603, "clip_ratio/low_min": 0.1614802423864603, "clip_ratio/high_mean": 0.05551740154623985, "clip_ratio/high_max": 0.05551740154623985, "clip_ratio/region_mean": 0.21699764393270016, "reward_total_mean": 0.2376668006181717, "reward_meter_mean": 0.3581885099411011, "reward_meter_std": 0.3801507353782654, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.993811845779419, "reward_repeat_soft_std": 0.005281467922031879, "reward_judge_quality_mean": 0.3799999952316284, "reward_judge_quality_std": 0.2112547606229782, "reward_total_composite_mean": 0.2376668006181717, "reward_total_composite_std": 0.22460903227329254} {"timestamp_utc": "2026-04-13T13:09:50Z", "mode": "train", "global_step": 199, "epoch": 0.019989954796584632, "loss": 0.3595, "grad_norm": 22.297731399536133, "learning_rate": 9.4e-06, "num_tokens": 353200.0, "completions/mean_length": 26.25, "completions/min_length": 16.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.25, "completions/min_terminated_length": 16.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.5058358907699585, "rewards/meter/std": 0.4252464175224304, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.5175492167472839, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9765625, "rewards/repeat_soft/std": 0.019408106803894043, "rewards/judge_quality/mean": 0.25999999046325684, "rewards/judge_quality/std": 0.06633248925209045, "rewards/total_composite/mean": 0.15322190523147583, "rewards/total_composite/std": 0.0743454247713089, "reward": 0.15322190523147583, "reward_std": 0.0743454247713089, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2133527547121048, "sampling/sampling_logp_difference/max": 0.93658447265625, "sampling/importance_sampling_ratio/min": 0.391964316368103, "sampling/importance_sampling_ratio/mean": 1.0662285089492798, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 3.532737761735916, "clip_ratio/low_mean": 0.08266637660562992, "clip_ratio/low_min": 0.08266637660562992, "clip_ratio/high_mean": 0.07500000111758709, "clip_ratio/high_max": 0.07500000111758709, "clip_ratio/region_mean": 0.157666377723217, "reward_total_mean": 0.15322190523147583, "reward_meter_mean": 0.5058358907699585, "reward_meter_std": 0.4252464175224304, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.5175492167472839, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9765625, "reward_repeat_soft_std": 0.019408106803894043, "reward_judge_quality_mean": 0.25999999046325684, "reward_judge_quality_std": 0.06633248925209045, "reward_total_composite_mean": 0.15322190523147583, "reward_total_composite_std": 0.0743454247713089} {"timestamp_utc": "2026-04-13T13:10:00Z", "mode": "train", "global_step": 200, "epoch": 0.020090406830738324, "loss": 0.731, "grad_norm": 10.606621742248535, "learning_rate": 9.396969696969697e-06, "num_tokens": 355647.0, "completions/mean_length": 131.875, "completions/min_length": 73.0, "completions/max_length": 451.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.875, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 451.0, "rewards/meter/mean": 0.7083677053451538, "rewards/meter/std": 0.39610084891319275, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.18898223340511322, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9976977705955505, "rewards/repeat_soft/std": 0.002274994971230626, "rewards/judge_quality/mean": 0.16124999523162842, "rewards/judge_quality/std": 0.05249149724841118, "rewards/total_composite/mean": 0.1362566351890564, "rewards/total_composite/std": 0.07544992119073868, "reward": 0.1362566351890564, "reward_std": 0.07544991374015808, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22925913333892822, "sampling/sampling_logp_difference/max": 1.3327980041503906, "sampling/importance_sampling_ratio/min": 0.2637382745742798, "sampling/importance_sampling_ratio/mean": 1.0662697553634644, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 5.329273879528046, "clip_ratio/low_mean": 0.04947514459490776, "clip_ratio/low_min": 0.04947514459490776, "clip_ratio/high_mean": 0.11767296493053436, "clip_ratio/high_max": 0.11767296493053436, "clip_ratio/region_mean": 0.16714810952544212, "reward_total_mean": 0.1362566351890564, "reward_meter_mean": 0.7083677053451538, "reward_meter_std": 0.39610084891319275, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.18898223340511322, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9976977705955505, "reward_repeat_soft_std": 0.002274994971230626, "reward_judge_quality_mean": 0.16124999523162842, "reward_judge_quality_std": 0.05249149724841118, "reward_total_composite_mean": 0.1362566351890564, "reward_total_composite_std": 0.07544992119073868} {"timestamp_utc": "2026-04-13T13:10:58Z", "mode": "eval", "global_step": 200, "epoch": 0.020090406830738324, "eval_loss": NaN, "eval_runtime": 58.0055, "eval_samples_per_second": 1.379, "eval_steps_per_second": 0.172, "eval_num_tokens": 355647.0, "eval_completions/mean_length": 97.65, "eval_completions/min_length": 27.2, "eval_completions/max_length": 271.4, "eval_completions/clipped_ratio": 0.0375, "eval_completions/mean_terminated_length": 83.01964492797852, "eval_completions/min_terminated_length": 27.2, "eval_completions/max_terminated_length": 213.9, "eval_rewards/meter/mean": 0.3731417119503021, "eval_rewards/meter/std": 0.32877095341682433, "eval_rewards/count_adherence/mean": 0.9172916769981384, "eval_rewards/count_adherence/std": 0.16572578102350236, "eval_rewards/hard_gate/mean": 0.85, "eval_rewards/hard_gate/std": 0.28198386132717135, "eval_rewards/repeat_soft/mean": 0.9944182991981506, "eval_rewards/repeat_soft/std": 0.008176561235450209, "eval_rewards/judge_quality/mean": 0.2240000009536743, "eval_rewards/judge_quality/std": 0.09404514394700528, "eval_rewards/total_composite/mean": 0.126047283411026, "eval_rewards/total_composite/std": 0.08404805548489094, "eval_reward": 0.126047283411026, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.1952221006155014, "eval_sampling/sampling_logp_difference/max": 1.064872646331787, "eval_sampling/importance_sampling_ratio/min": 0.34764412939548495, "eval_sampling/importance_sampling_ratio/mean": 1.060666012763977, "eval_sampling/importance_sampling_ratio/max": 1.7729849934577941, "eval_entropy": 4.640206694602966, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.126047283411026, "eval_reward_meter_mean": 0.3731417119503021, "eval_reward_meter_std": 0.32877095341682433, "eval_reward_count_adherence_mean": 0.9172916769981384, "eval_reward_count_adherence_std": 0.16572578102350236, "eval_reward_hard_gate_mean": 0.85, "eval_reward_hard_gate_std": 0.28198386132717135, "eval_reward_repeat_soft_mean": 0.9944182991981506, "eval_reward_repeat_soft_std": 0.008176561235450209, "eval_reward_judge_quality_mean": 0.2240000009536743, "eval_reward_judge_quality_std": 0.09404514394700528, "eval_reward_total_composite_mean": 0.126047283411026, "eval_reward_total_composite_std": 0.08404805548489094}