diff --git "a/metrics.jsonl" "b/metrics.jsonl" --- "a/metrics.jsonl" +++ "b/metrics.jsonl" @@ -151,3 +151,55 @@ {"timestamp_utc": "2026-04-12T22:34:33Z", "mode": "train", "global_step": 149, "epoch": 0.01496735308890005, "loss": 0.0005, "grad_norm": 5.917326927185059, "learning_rate": 9.551515151515152e-06, "num_tokens": 289547.0, "completions/mean_length": 196.125, "completions/min_length": 170.0, "completions/max_length": 207.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 196.125, "completions/min_terminated_length": 170.0, "completions/max_terminated_length": 207.0, "rewards/meter/mean": 0.9900409579277039, "rewards/meter/std": 0.01028269249945879, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9094256162643433, "rewards/repeat_soft/std": 0.09267400950193405, "rewards/judge_quality/mean": 0.2212499976158142, "rewards/judge_quality/std": 0.09433034062385559, "rewards/total_composite/mean": 0.657077431678772, "rewards/total_composite/std": 0.2678874731063843, "reward": 0.657077431678772, "reward_std": 0.2678874731063843, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17067110538482666, "sampling/sampling_logp_difference/max": 1.832137107849121, "sampling/importance_sampling_ratio/min": 0.16007111966609955, "sampling/importance_sampling_ratio/mean": 1.0396323204040527, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1036142259836197, "clip_ratio/low_mean": 0.018518518656492233, "clip_ratio/low_min": 0.018518518656492233, "clip_ratio/high_mean": 0.12627116777002811, "clip_ratio/high_max": 0.12627116777002811, "clip_ratio/region_mean": 0.14478968642652035, "reward_total_mean": 0.657077431678772, "reward_meter_mean": 0.9900409579277039, "reward_meter_std": 0.01028269249945879, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9094256162643433, "reward_repeat_soft_std": 0.09267400950193405, "reward_judge_quality_mean": 0.2212499976158142, "reward_judge_quality_std": 0.09433034062385559, "reward_total_composite_mean": 0.657077431678772, "reward_total_composite_std": 0.2678874731063843} {"timestamp_utc": "2026-04-12T22:34:41Z", "mode": "train", "global_step": 150, "epoch": 0.015067805123053743, "loss": 0.0307, "grad_norm": 8.67408275604248, "learning_rate": 9.54848484848485e-06, "num_tokens": 291984.0, "completions/mean_length": 134.625, "completions/min_length": 114.0, "completions/max_length": 164.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 134.625, "completions/min_terminated_length": 114.0, "completions/max_terminated_length": 164.0, "rewards/meter/mean": 0.6442027688026428, "rewards/meter/std": 0.23864556849002838, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.1035098284482956, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9940277338027954, "rewards/repeat_soft/std": 0.0051259868778288364, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.641294002532959, "rewards/total_composite/std": 0.1019832044839859, "reward": 0.641294002532959, "reward_std": 0.1019832044839859, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19296503067016602, "sampling/sampling_logp_difference/max": 1.9423902034759521, "sampling/importance_sampling_ratio/min": 0.14336088299751282, "sampling/importance_sampling_ratio/mean": 1.0387064218521118, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8878242373466492, "clip_ratio/low_mean": 0.12727568112313747, "clip_ratio/low_min": 0.12727568112313747, "clip_ratio/high_mean": 0.06800262071192265, "clip_ratio/high_max": 0.06800262071192265, "clip_ratio/region_mean": 0.19527830183506012, "reward_total_mean": 0.641294002532959, "reward_meter_mean": 0.6442027688026428, "reward_meter_std": 0.23864556849002838, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.1035098284482956, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9940277338027954, "reward_repeat_soft_std": 0.0051259868778288364, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.641294002532959, "reward_total_composite_std": 0.1019832044839859} {"timestamp_utc": "2026-04-12T22:35:28Z", "mode": "eval", "global_step": 150, "epoch": 0.015067805123053743, "eval_loss": NaN, "eval_runtime": 47.3495, "eval_samples_per_second": 1.69, "eval_steps_per_second": 0.211, "eval_num_tokens": 291984.0, "eval_completions/mean_length": 89.7, "eval_completions/min_length": 39.4, "eval_completions/max_length": 155.8, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 89.7, "eval_completions/min_terminated_length": 39.4, "eval_completions/max_terminated_length": 155.8, "eval_rewards/meter/mean": 0.579559576511383, "eval_rewards/meter/std": 0.3873100936412811, "eval_rewards/count_adherence/mean": 0.9793750047683716, "eval_rewards/count_adherence/std": 0.046562766283750535, "eval_rewards/hard_gate/mean": 0.975, "eval_rewards/hard_gate/std": 0.07071067690849304, "eval_rewards/repeat_soft/mean": 0.9804000198841095, "eval_rewards/repeat_soft/std": 0.034216074645519255, "eval_rewards/judge_quality/mean": 0.38399999141693114, "eval_rewards/judge_quality/std": 0.13652184307575227, "eval_rewards/total_composite/mean": 0.6085451781749726, "eval_rewards/total_composite/std": 0.1927956983447075, "eval_reward": 0.6085451781749726, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.13093501552939416, "eval_sampling/sampling_logp_difference/max": 1.1314985275268554, "eval_sampling/importance_sampling_ratio/min": 0.3271831452846527, "eval_sampling/importance_sampling_ratio/mean": 1.0350063562393188, "eval_sampling/importance_sampling_ratio/max": 1.5271484375, "eval_entropy": 1.9806707382202149, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6085451781749726, "eval_reward_meter_mean": 0.579559576511383, "eval_reward_meter_std": 0.3873100936412811, "eval_reward_count_adherence_mean": 0.9793750047683716, "eval_reward_count_adherence_std": 0.046562766283750535, "eval_reward_hard_gate_mean": 0.975, "eval_reward_hard_gate_std": 0.07071067690849304, "eval_reward_repeat_soft_mean": 0.9804000198841095, "eval_reward_repeat_soft_std": 0.034216074645519255, "eval_reward_judge_quality_mean": 0.38399999141693114, "eval_reward_judge_quality_std": 0.13652184307575227, "eval_reward_total_composite_mean": 0.6085451781749726, "eval_reward_total_composite_std": 0.1927956983447075} +{"timestamp_utc": "2026-04-12T22:35:40Z", "mode": "train", "global_step": 151, "epoch": 0.015168257157207434, "loss": -0.0268, "grad_norm": 17.52226448059082, "learning_rate": 9.545454545454547e-06, "num_tokens": 293438.0, "completions/mean_length": 29.75, "completions/min_length": 27.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.75, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.6037045121192932, "rewards/meter/std": 0.4519675076007843, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.960254430770874, "rewards/repeat_soft/std": 0.005280489567667246, "rewards/judge_quality/mean": 0.3687499761581421, "rewards/judge_quality/std": 0.11319231241941452, "rewards/total_composite/mean": 0.6283174753189087, "rewards/total_composite/std": 0.20097284018993378, "reward": 0.6283174753189087, "reward_std": 0.20097284018993378, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2218317985534668, "sampling/sampling_logp_difference/max": 1.4685611724853516, "sampling/importance_sampling_ratio/min": 0.23025654256343842, "sampling/importance_sampling_ratio/mean": 1.0167431831359863, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6521061211824417, "clip_ratio/low_mean": 0.08347017224878073, "clip_ratio/low_min": 0.08347017224878073, "clip_ratio/high_mean": 0.1568104773759842, "clip_ratio/high_max": 0.1568104773759842, "clip_ratio/region_mean": 0.24028064962476492, "reward_total_mean": 0.6283174753189087, "reward_meter_mean": 0.6037045121192932, "reward_meter_std": 0.4519675076007843, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.960254430770874, "reward_repeat_soft_std": 0.005280489567667246, "reward_judge_quality_mean": 0.3687499761581421, "reward_judge_quality_std": 0.11319231241941452, "reward_total_composite_mean": 0.6283174753189087, "reward_total_composite_std": 0.20097284018993378} +{"timestamp_utc": "2026-04-12T22:35:47Z", "mode": "train", "global_step": 152, "epoch": 0.015268709191361125, "loss": -0.0616, "grad_norm": 7.388057231903076, "learning_rate": 9.542424242424242e-06, "num_tokens": 295656.0, "completions/mean_length": 114.25, "completions/min_length": 69.0, "completions/max_length": 139.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 114.25, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 139.0, "rewards/meter/mean": 0.8712646961212158, "rewards/meter/std": 0.28484392166137695, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9923064708709717, "rewards/repeat_soft/std": 0.010224386118352413, "rewards/judge_quality/mean": 0.2887499928474426, "rewards/judge_quality/std": 0.11630470305681229, "rewards/total_composite/mean": 0.7232372164726257, "rewards/total_composite/std": 0.11754972487688065, "reward": 0.7232372164726257, "reward_std": 0.11754971742630005, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19024528563022614, "sampling/sampling_logp_difference/max": 1.115931510925293, "sampling/importance_sampling_ratio/min": 0.32760995626449585, "sampling/importance_sampling_ratio/mean": 1.0422486066818237, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.7057955861091614, "clip_ratio/low_mean": 0.05840101931244135, "clip_ratio/low_min": 0.05840101931244135, "clip_ratio/high_mean": 0.11011046543717384, "clip_ratio/high_max": 0.11011046543717384, "clip_ratio/region_mean": 0.1685114847496152, "reward_total_mean": 0.7232372164726257, "reward_meter_mean": 0.8712646961212158, "reward_meter_std": 0.28484392166137695, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9923064708709717, "reward_repeat_soft_std": 0.010224386118352413, "reward_judge_quality_mean": 0.2887499928474426, "reward_judge_quality_std": 0.11630470305681229, "reward_total_composite_mean": 0.7232372164726257, "reward_total_composite_std": 0.11754972487688065} +{"timestamp_utc": "2026-04-12T22:35:58Z", "mode": "train", "global_step": 153, "epoch": 0.015369161225514816, "loss": 0.0206, "grad_norm": 14.170472145080566, "learning_rate": 9.539393939393941e-06, "num_tokens": 297271.0, "completions/mean_length": 55.875, "completions/min_length": 47.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.875, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9694111347198486, "rewards/meter/std": 0.035776134580373764, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9994534254074097, "rewards/repeat_soft/std": 0.001545967417769134, "rewards/judge_quality/mean": 0.44999998807907104, "rewards/judge_quality/std": 0.0, "rewards/total_composite/mean": 0.7175350785255432, "rewards/total_composite/std": 0.29035794734954834, "reward": 0.7175350785255432, "reward_std": 0.29035791754722595, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17416347563266754, "sampling/sampling_logp_difference/max": 1.2880401611328125, "sampling/importance_sampling_ratio/min": 0.2758108079433441, "sampling/importance_sampling_ratio/mean": 1.0363378524780273, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8942160308361053, "clip_ratio/low_mean": 0.02801724150776863, "clip_ratio/low_min": 0.02801724150776863, "clip_ratio/high_mean": 0.13300238270312548, "clip_ratio/high_max": 0.13300238270312548, "clip_ratio/region_mean": 0.1610196242108941, "reward_total_mean": 0.7175350785255432, "reward_meter_mean": 0.9694111347198486, "reward_meter_std": 0.035776134580373764, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9994534254074097, "reward_repeat_soft_std": 0.001545967417769134, "reward_judge_quality_mean": 0.44999998807907104, "reward_judge_quality_std": 0.0, "reward_total_composite_mean": 0.7175350785255432, "reward_total_composite_std": 0.29035794734954834} +{"timestamp_utc": "2026-04-12T22:36:09Z", "mode": "train", "global_step": 154, "epoch": 0.015469613259668509, "loss": 0.0062, "grad_norm": 14.80764102935791, "learning_rate": 9.536363636363637e-06, "num_tokens": 299067.0, "completions/mean_length": 52.5, "completions/min_length": 36.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.5, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.7321591377258301, "rewards/meter/std": 0.37825655937194824, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 1.0, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.3675000071525574, "rewards/judge_quality/std": 0.09808888286352158, "rewards/total_composite/mean": 0.6803466081619263, "rewards/total_composite/std": 0.18387891352176666, "reward": 0.6803466081619263, "reward_std": 0.18387891352176666, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2041114717721939, "sampling/sampling_logp_difference/max": 1.7619848251342773, "sampling/importance_sampling_ratio/min": 0.17170371115207672, "sampling/importance_sampling_ratio/mean": 1.0717298984527588, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.875928968191147, "clip_ratio/low_mean": 0.055980125442147255, "clip_ratio/low_min": 0.055980125442147255, "clip_ratio/high_mean": 0.1288575418293476, "clip_ratio/high_max": 0.1288575418293476, "clip_ratio/region_mean": 0.18483766727149487, "reward_total_mean": 0.6803466081619263, "reward_meter_mean": 0.7321591377258301, "reward_meter_std": 0.37825655937194824, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 1.0, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.3675000071525574, "reward_judge_quality_std": 0.09808888286352158, "reward_total_composite_mean": 0.6803466081619263, "reward_total_composite_std": 0.18387891352176666} +{"timestamp_utc": "2026-04-12T22:36:17Z", "mode": "train", "global_step": 155, "epoch": 0.0155700652938222, "loss": 0.0592, "grad_norm": 10.283768653869629, "learning_rate": 9.533333333333334e-06, "num_tokens": 300993.0, "completions/mean_length": 64.75, "completions/min_length": 52.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.75, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.2703203856945038, "rewards/meter/std": 0.4463132321834564, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9945789575576782, "rewards/repeat_soft/std": 0.0068430774845182896, "rewards/judge_quality/mean": 0.4124999940395355, "rewards/judge_quality/std": 0.1685018092393875, "rewards/total_composite/mean": 0.49485206604003906, "rewards/total_composite/std": 0.21574346721172333, "reward": 0.49485206604003906, "reward_std": 0.21574346721172333, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2035830020904541, "sampling/sampling_logp_difference/max": 1.513336181640625, "sampling/importance_sampling_ratio/min": 0.22017420828342438, "sampling/importance_sampling_ratio/mean": 1.0503827333450317, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3192920982837677, "clip_ratio/low_mean": 0.10545289795845747, "clip_ratio/low_min": 0.10545289795845747, "clip_ratio/high_mean": 0.04740618169307709, "clip_ratio/high_max": 0.04740618169307709, "clip_ratio/region_mean": 0.15285907965153456, "reward_total_mean": 0.49485206604003906, "reward_meter_mean": 0.2703203856945038, "reward_meter_std": 0.4463132321834564, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9945789575576782, "reward_repeat_soft_std": 0.0068430774845182896, "reward_judge_quality_mean": 0.4124999940395355, "reward_judge_quality_std": 0.1685018092393875, "reward_total_composite_mean": 0.49485206604003906, "reward_total_composite_std": 0.21574346721172333} +{"timestamp_utc": "2026-04-12T22:36:24Z", "mode": "train", "global_step": 156, "epoch": 0.01567051732797589, "loss": -0.0075, "grad_norm": 16.434175491333008, "learning_rate": 9.530303030303031e-06, "num_tokens": 302725.0, "completions/mean_length": 53.5, "completions/min_length": 42.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.5, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.6622544527053833, "rewards/meter/std": 0.35615384578704834, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9921135306358337, "rewards/repeat_soft/std": 0.009797083213925362, "rewards/judge_quality/mean": 0.5012500286102295, "rewards/judge_quality/std": 0.16974246501922607, "rewards/total_composite/mean": 0.6976008415222168, "rewards/total_composite/std": 0.1817840188741684, "reward": 0.6976008415222168, "reward_std": 0.1817840188741684, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16653680801391602, "sampling/sampling_logp_difference/max": 1.999290943145752, "sampling/importance_sampling_ratio/min": 0.13543127477169037, "sampling/importance_sampling_ratio/mean": 1.0143111944198608, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.441454216837883, "clip_ratio/low_mean": 0.07074090465903282, "clip_ratio/low_min": 0.07074090465903282, "clip_ratio/high_mean": 0.08608901780098677, "clip_ratio/high_max": 0.08608901780098677, "clip_ratio/region_mean": 0.1568299224600196, "reward_total_mean": 0.6976008415222168, "reward_meter_mean": 0.6622544527053833, "reward_meter_std": 0.35615384578704834, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9921135306358337, "reward_repeat_soft_std": 0.009797083213925362, "reward_judge_quality_mean": 0.5012500286102295, "reward_judge_quality_std": 0.16974246501922607, "reward_total_composite_mean": 0.6976008415222168, "reward_total_composite_std": 0.1817840188741684} +{"timestamp_utc": "2026-04-12T22:36:31Z", "mode": "train", "global_step": 157, "epoch": 0.015770969362129583, "loss": 0.0603, "grad_norm": 20.17525291442871, "learning_rate": 9.527272727272729e-06, "num_tokens": 304186.0, "completions/mean_length": 29.625, "completions/min_length": 25.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.625, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.6452634334564209, "rewards/meter/std": 0.4738078713417053, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9538979530334473, "rewards/repeat_soft/std": 0.024330154061317444, "rewards/judge_quality/mean": 0.2800000011920929, "rewards/judge_quality/std": 0.13125765323638916, "rewards/total_composite/mean": 0.6197583675384521, "rewards/total_composite/std": 0.21760250627994537, "reward": 0.6197583675384521, "reward_std": 0.21760250627994537, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19076097011566162, "sampling/sampling_logp_difference/max": 1.256563663482666, "sampling/importance_sampling_ratio/min": 0.2846304476261139, "sampling/importance_sampling_ratio/mean": 1.0374767780303955, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8032512366771698, "clip_ratio/low_mean": 0.07534630037844181, "clip_ratio/low_min": 0.07534630037844181, "clip_ratio/high_mean": 0.11880760360509157, "clip_ratio/high_max": 0.11880760360509157, "clip_ratio/region_mean": 0.19415390398353338, "reward_total_mean": 0.6197583675384521, "reward_meter_mean": 0.6452634334564209, "reward_meter_std": 0.4738078713417053, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9538979530334473, "reward_repeat_soft_std": 0.024330154061317444, "reward_judge_quality_mean": 0.2800000011920929, "reward_judge_quality_std": 0.13125765323638916, "reward_total_composite_mean": 0.6197583675384521, "reward_total_composite_std": 0.21760250627994537} +{"timestamp_utc": "2026-04-12T22:36:38Z", "mode": "train", "global_step": 158, "epoch": 0.015871421396283274, "loss": 0.0186, "grad_norm": 8.423430442810059, "learning_rate": 9.524242424242424e-06, "num_tokens": 305938.0, "completions/mean_length": 71.0, "completions/min_length": 59.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.0, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9905709028244019, "rewards/meter/std": 0.013387288898229599, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9690340161323547, "rewards/repeat_soft/std": 0.03389734774827957, "rewards/judge_quality/mean": 0.36000001430511475, "rewards/judge_quality/std": 0.09165151417255402, "rewards/total_composite/mean": 0.800660252571106, "rewards/total_composite/std": 0.03153316676616669, "reward": 0.800660252571106, "reward_std": 0.03153315186500549, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16707605123519897, "sampling/sampling_logp_difference/max": 1.6637210845947266, "sampling/importance_sampling_ratio/min": 0.1894327700138092, "sampling/importance_sampling_ratio/mean": 1.036516547203064, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6277482360601425, "clip_ratio/low_mean": 0.05309281498193741, "clip_ratio/low_min": 0.05309281498193741, "clip_ratio/high_mean": 0.09371021948754787, "clip_ratio/high_max": 0.09371021948754787, "clip_ratio/region_mean": 0.14680303446948528, "reward_total_mean": 0.800660252571106, "reward_meter_mean": 0.9905709028244019, "reward_meter_std": 0.013387288898229599, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9690340161323547, "reward_repeat_soft_std": 0.03389734774827957, "reward_judge_quality_mean": 0.36000001430511475, "reward_judge_quality_std": 0.09165151417255402, "reward_total_composite_mean": 0.800660252571106, "reward_total_composite_std": 0.03153316676616669} +{"timestamp_utc": "2026-04-12T22:36:47Z", "mode": "train", "global_step": 159, "epoch": 0.015971873430436965, "loss": -0.085, "grad_norm": 14.062602996826172, "learning_rate": 9.521212121212121e-06, "num_tokens": 307660.0, "completions/mean_length": 62.25, "completions/min_length": 45.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.25, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9907737374305725, "rewards/meter/std": 0.008397881872951984, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9953635931015015, "rewards/repeat_soft/std": 0.004471874330192804, "rewards/judge_quality/mean": 0.44624999165534973, "rewards/judge_quality/std": 0.010606602765619755, "rewards/total_composite/mean": 0.8292595744132996, "rewards/total_composite/std": 0.00623986916616559, "reward": 0.8292595744132996, "reward_std": 0.00623986916616559, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17575737833976746, "sampling/sampling_logp_difference/max": 1.4365453720092773, "sampling/importance_sampling_ratio/min": 0.2377476692199707, "sampling/importance_sampling_ratio/mean": 1.0418425798416138, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.062304139137268, "clip_ratio/low_mean": 0.05505712330341339, "clip_ratio/low_min": 0.05505712330341339, "clip_ratio/high_mean": 0.09878544509410858, "clip_ratio/high_max": 0.09878544509410858, "clip_ratio/region_mean": 0.15384256839752197, "reward_total_mean": 0.8292595744132996, "reward_meter_mean": 0.9907737374305725, "reward_meter_std": 0.008397881872951984, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9953635931015015, "reward_repeat_soft_std": 0.004471874330192804, "reward_judge_quality_mean": 0.44624999165534973, "reward_judge_quality_std": 0.010606602765619755, "reward_total_composite_mean": 0.8292595744132996, "reward_total_composite_std": 0.00623986916616559} +{"timestamp_utc": "2026-04-12T22:36:54Z", "mode": "train", "global_step": 160, "epoch": 0.01607232546459066, "loss": 0.1267, "grad_norm": 16.92506980895996, "learning_rate": 9.518181818181819e-06, "num_tokens": 309267.0, "completions/mean_length": 43.875, "completions/min_length": 28.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.875, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9226127862930298, "rewards/meter/std": 0.09503024071455002, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9710944890975952, "rewards/repeat_soft/std": 0.05597909912467003, "rewards/judge_quality/mean": 0.6949999928474426, "rewards/judge_quality/std": 0.1908627152442932, "rewards/total_composite/mean": 0.8707852363586426, "rewards/total_composite/std": 0.05670774728059769, "reward": 0.8707852363586426, "reward_std": 0.056707751005887985, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16918140649795532, "sampling/sampling_logp_difference/max": 1.366260051727295, "sampling/importance_sampling_ratio/min": 0.2550590932369232, "sampling/importance_sampling_ratio/mean": 1.0055161714553833, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.5125083029270172, "clip_ratio/low_mean": 0.04656171798706055, "clip_ratio/low_min": 0.04656171798706055, "clip_ratio/high_mean": 0.08021787973120809, "clip_ratio/high_max": 0.08021787973120809, "clip_ratio/region_mean": 0.12677959771826863, "reward_total_mean": 0.8707852363586426, "reward_meter_mean": 0.9226127862930298, "reward_meter_std": 0.09503024071455002, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9710944890975952, "reward_repeat_soft_std": 0.05597909912467003, "reward_judge_quality_mean": 0.6949999928474426, "reward_judge_quality_std": 0.1908627152442932, "reward_total_composite_mean": 0.8707852363586426, "reward_total_composite_std": 0.05670774728059769} +{"timestamp_utc": "2026-04-12T22:37:02Z", "mode": "train", "global_step": 161, "epoch": 0.01617277749874435, "loss": -0.0626, "grad_norm": 15.525965690612793, "learning_rate": 9.515151515151516e-06, "num_tokens": 310703.0, "completions/mean_length": 33.5, "completions/min_length": 27.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.5, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.8751624822616577, "rewards/meter/std": 0.3393276631832123, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9143266677856445, "rewards/repeat_soft/std": 0.10079991817474365, "rewards/judge_quality/mean": 0.4975000321865082, "rewards/judge_quality/std": 0.2817166745662689, "rewards/total_composite/mean": 0.7845057845115662, "rewards/total_composite/std": 0.20682775974273682, "reward": 0.7845057845115662, "reward_std": 0.20682775974273682, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17265181243419647, "sampling/sampling_logp_difference/max": 1.8043546676635742, "sampling/importance_sampling_ratio/min": 0.16458062827587128, "sampling/importance_sampling_ratio/mean": 1.0136698484420776, "sampling/importance_sampling_ratio/max": 1.8944414854049683, "entropy": 1.8272867798805237, "clip_ratio/low_mean": 0.0351851861923933, "clip_ratio/low_min": 0.0351851861923933, "clip_ratio/high_mean": 0.12637826520949602, "clip_ratio/high_max": 0.12637826520949602, "clip_ratio/region_mean": 0.16156345140188932, "reward_total_mean": 0.7845057845115662, "reward_meter_mean": 0.8751624822616577, "reward_meter_std": 0.3393276631832123, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9143266677856445, "reward_repeat_soft_std": 0.10079991817474365, "reward_judge_quality_mean": 0.4975000321865082, "reward_judge_quality_std": 0.2817166745662689, "reward_total_composite_mean": 0.7845057845115662, "reward_total_composite_std": 0.20682775974273682} +{"timestamp_utc": "2026-04-12T22:37:11Z", "mode": "train", "global_step": 162, "epoch": 0.016273229532898042, "loss": 0.0382, "grad_norm": 19.982229232788086, "learning_rate": 9.512121212121213e-06, "num_tokens": 312289.0, "completions/mean_length": 38.25, "completions/min_length": 28.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.25, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.5322328805923462, "rewards/meter/std": 0.3870828151702881, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9937243461608887, "rewards/repeat_soft/std": 0.009791112504899502, "rewards/judge_quality/mean": 0.5349999666213989, "rewards/judge_quality/std": 0.24663449823856354, "rewards/total_composite/mean": 0.6493772268295288, "rewards/total_composite/std": 0.2318180650472641, "reward": 0.6493772268295288, "reward_std": 0.2318180501461029, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22324085235595703, "sampling/sampling_logp_difference/max": 2.7204360961914062, "sampling/importance_sampling_ratio/min": 0.0658460333943367, "sampling/importance_sampling_ratio/mean": 1.00266432762146, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6748038977384567, "clip_ratio/low_mean": 0.09446094930171967, "clip_ratio/low_min": 0.09446094930171967, "clip_ratio/high_mean": 0.0970759429037571, "clip_ratio/high_max": 0.0970759429037571, "clip_ratio/region_mean": 0.19153689220547676, "reward_total_mean": 0.6493772268295288, "reward_meter_mean": 0.5322328805923462, "reward_meter_std": 0.3870828151702881, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9937243461608887, "reward_repeat_soft_std": 0.009791112504899502, "reward_judge_quality_mean": 0.5349999666213989, "reward_judge_quality_std": 0.24663449823856354, "reward_total_composite_mean": 0.6493772268295288, "reward_total_composite_std": 0.2318180650472641} +{"timestamp_utc": "2026-04-12T22:37:19Z", "mode": "train", "global_step": 163, "epoch": 0.016373681567051733, "loss": 0.016, "grad_norm": 11.819523811340332, "learning_rate": 9.50909090909091e-06, "num_tokens": 314068.0, "completions/mean_length": 62.375, "completions/min_length": 57.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.375, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9233949780464172, "rewards/meter/std": 0.19471479952335358, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9969461560249329, "rewards/repeat_soft/std": 0.004109133966267109, "rewards/judge_quality/mean": 0.42624998092651367, "rewards/judge_quality/std": 0.14647647738456726, "rewards/total_composite/mean": 0.7930973768234253, "rewards/total_composite/std": 0.11556258052587509, "reward": 0.7930973768234253, "reward_std": 0.1155625730752945, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17588582634925842, "sampling/sampling_logp_difference/max": 1.965134620666504, "sampling/importance_sampling_ratio/min": 0.14013701677322388, "sampling/importance_sampling_ratio/mean": 1.0216656923294067, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.6963207721710205, "clip_ratio/low_mean": 0.0425052959471941, "clip_ratio/low_min": 0.0425052959471941, "clip_ratio/high_mean": 0.1385966893285513, "clip_ratio/high_max": 0.1385966893285513, "clip_ratio/region_mean": 0.1811019852757454, "reward_total_mean": 0.7930973768234253, "reward_meter_mean": 0.9233949780464172, "reward_meter_std": 0.19471479952335358, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9969461560249329, "reward_repeat_soft_std": 0.004109133966267109, "reward_judge_quality_mean": 0.42624998092651367, "reward_judge_quality_std": 0.14647647738456726, "reward_total_composite_mean": 0.7930973768234253, "reward_total_composite_std": 0.11556258052587509} +{"timestamp_utc": "2026-04-12T22:37:28Z", "mode": "train", "global_step": 164, "epoch": 0.016474133601205424, "loss": 0.0119, "grad_norm": 8.069746971130371, "learning_rate": 9.506060606060606e-06, "num_tokens": 316596.0, "completions/mean_length": 129.0, "completions/min_length": 116.0, "completions/max_length": 139.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 129.0, "completions/min_terminated_length": 116.0, "completions/max_terminated_length": 139.0, "rewards/meter/mean": 0.6153008937835693, "rewards/meter/std": 0.3726039528846741, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9899668097496033, "rewards/repeat_soft/std": 0.012333144433796406, "rewards/judge_quality/mean": 0.32249999046325684, "rewards/judge_quality/std": 0.10925068706274033, "rewards/total_composite/mean": 0.6226321458816528, "rewards/total_composite/std": 0.19220340251922607, "reward": 0.6226321458816528, "reward_std": 0.19220338761806488, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2005292922258377, "sampling/sampling_logp_difference/max": 1.548935890197754, "sampling/importance_sampling_ratio/min": 0.21247395873069763, "sampling/importance_sampling_ratio/mean": 1.0147041082382202, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.309463620185852, "clip_ratio/low_mean": 0.076393430121243, "clip_ratio/low_min": 0.076393430121243, "clip_ratio/high_mean": 0.09089494496583939, "clip_ratio/high_max": 0.09089494496583939, "clip_ratio/region_mean": 0.16728837508708239, "reward_total_mean": 0.6226321458816528, "reward_meter_mean": 0.6153008937835693, "reward_meter_std": 0.3726039528846741, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9899668097496033, "reward_repeat_soft_std": 0.012333144433796406, "reward_judge_quality_mean": 0.32249999046325684, "reward_judge_quality_std": 0.10925068706274033, "reward_total_composite_mean": 0.6226321458816528, "reward_total_composite_std": 0.19220340251922607} +{"timestamp_utc": "2026-04-12T22:37:35Z", "mode": "train", "global_step": 165, "epoch": 0.016574585635359115, "loss": 0.0402, "grad_norm": 15.899052619934082, "learning_rate": 9.503030303030303e-06, "num_tokens": 318202.0, "completions/mean_length": 42.75, "completions/min_length": 38.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.75, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.5030492544174194, "rewards/meter/std": 0.417573481798172, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.991955041885376, "rewards/repeat_soft/std": 0.00785687007009983, "rewards/judge_quality/mean": 0.6075000166893005, "rewards/judge_quality/std": 0.25877460837364197, "rewards/total_composite/mean": 0.6578176617622375, "rewards/total_composite/std": 0.18415668606758118, "reward": 0.6578176617622375, "reward_std": 0.18415668606758118, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1723388433456421, "sampling/sampling_logp_difference/max": 1.7610225677490234, "sampling/importance_sampling_ratio/min": 0.17186902463436127, "sampling/importance_sampling_ratio/mean": 0.9937219619750977, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.7180194333195686, "clip_ratio/low_mean": 0.08987216651439667, "clip_ratio/low_min": 0.08987216651439667, "clip_ratio/high_mean": 0.08828502427786589, "clip_ratio/high_max": 0.08828502427786589, "clip_ratio/region_mean": 0.17815719079226255, "reward_total_mean": 0.6578176617622375, "reward_meter_mean": 0.5030492544174194, "reward_meter_std": 0.417573481798172, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.991955041885376, "reward_repeat_soft_std": 0.00785687007009983, "reward_judge_quality_mean": 0.6075000166893005, "reward_judge_quality_std": 0.25877460837364197, "reward_total_composite_mean": 0.6578176617622375, "reward_total_composite_std": 0.18415668606758118} +{"timestamp_utc": "2026-04-12T22:37:41Z", "mode": "train", "global_step": 166, "epoch": 0.016675037669512806, "loss": 0.0938, "grad_norm": 13.500005722045898, "learning_rate": 9.5e-06, "num_tokens": 319982.0, "completions/mean_length": 58.5, "completions/min_length": 52.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.5, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.5016564130783081, "rewards/meter/std": 0.40433311462402344, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9944120645523071, "rewards/repeat_soft/std": 0.0066553642973303795, "rewards/judge_quality/mean": 0.4099999964237213, "rewards/judge_quality/std": 0.06633248925209045, "rewards/total_composite/mean": 0.5981866121292114, "rewards/total_composite/std": 0.17862769961357117, "reward": 0.5981866121292114, "reward_std": 0.17862768471240997, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1928389072418213, "sampling/sampling_logp_difference/max": 1.3699474334716797, "sampling/importance_sampling_ratio/min": 0.2541202902793884, "sampling/importance_sampling_ratio/mean": 0.9989068508148193, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.914685234427452, "clip_ratio/low_mean": 0.0708869006484747, "clip_ratio/low_min": 0.0708869006484747, "clip_ratio/high_mean": 0.10799220204353333, "clip_ratio/high_max": 0.10799220204353333, "clip_ratio/region_mean": 0.17887910269200802, "reward_total_mean": 0.5981866121292114, "reward_meter_mean": 0.5016564130783081, "reward_meter_std": 0.40433311462402344, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9944120645523071, "reward_repeat_soft_std": 0.0066553642973303795, "reward_judge_quality_mean": 0.4099999964237213, "reward_judge_quality_std": 0.06633248925209045, "reward_total_composite_mean": 0.5981866121292114, "reward_total_composite_std": 0.17862769961357117} +{"timestamp_utc": "2026-04-12T22:37:49Z", "mode": "train", "global_step": 167, "epoch": 0.016775489703666498, "loss": -0.081, "grad_norm": 11.224985122680664, "learning_rate": 9.496969696969698e-06, "num_tokens": 321684.0, "completions/mean_length": 65.75, "completions/min_length": 45.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.75, "completions/min_terminated_length": 45.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9555763006210327, "rewards/meter/std": 0.06912393122911453, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9968563318252563, "rewards/repeat_soft/std": 0.006129227578639984, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.8079450130462646, "rewards/total_composite/std": 0.031966518610715866, "reward": 0.8079450130462646, "reward_std": 0.03196650743484497, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16898435354232788, "sampling/sampling_logp_difference/max": 1.017256736755371, "sampling/importance_sampling_ratio/min": 0.36158549785614014, "sampling/importance_sampling_ratio/mean": 1.0499838590621948, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.367111414670944, "clip_ratio/low_mean": 0.02604166604578495, "clip_ratio/low_min": 0.02604166604578495, "clip_ratio/high_mean": 0.1444350564852357, "clip_ratio/high_max": 0.1444350564852357, "clip_ratio/region_mean": 0.17047672253102064, "reward_total_mean": 0.8079450130462646, "reward_meter_mean": 0.9555763006210327, "reward_meter_std": 0.06912393122911453, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9968563318252563, "reward_repeat_soft_std": 0.006129227578639984, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.8079450130462646, "reward_total_composite_std": 0.031966518610715866} +{"timestamp_utc": "2026-04-12T22:37:57Z", "mode": "train", "global_step": 168, "epoch": 0.016875941737820192, "loss": 0.0637, "grad_norm": 18.318450927734375, "learning_rate": 9.493939393939395e-06, "num_tokens": 323245.0, "completions/mean_length": 30.125, "completions/min_length": 25.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.125, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.7876026630401611, "rewards/meter/std": 0.35172486305236816, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9564294219017029, "rewards/repeat_soft/std": 0.017170162871479988, "rewards/judge_quality/mean": 0.1875, "rewards/judge_quality/std": 0.0517549142241478, "rewards/total_composite/mean": 0.6563141345977783, "rewards/total_composite/std": 0.15374185144901276, "reward": 0.6563141345977783, "reward_std": 0.15374185144901276, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20536887645721436, "sampling/sampling_logp_difference/max": 1.2958226203918457, "sampling/importance_sampling_ratio/min": 0.2736726403236389, "sampling/importance_sampling_ratio/mean": 1.0338667631149292, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9300547987222672, "clip_ratio/low_mean": 0.04375000111758709, "clip_ratio/low_min": 0.04375000111758709, "clip_ratio/high_mean": 0.1926663313060999, "clip_ratio/high_max": 0.1926663313060999, "clip_ratio/region_mean": 0.23641633242368698, "reward_total_mean": 0.6563141345977783, "reward_meter_mean": 0.7876026630401611, "reward_meter_std": 0.35172486305236816, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9564294219017029, "reward_repeat_soft_std": 0.017170162871479988, "reward_judge_quality_mean": 0.1875, "reward_judge_quality_std": 0.0517549142241478, "reward_total_composite_mean": 0.6563141345977783, "reward_total_composite_std": 0.15374185144901276} +{"timestamp_utc": "2026-04-12T22:38:04Z", "mode": "train", "global_step": 169, "epoch": 0.016976393771973883, "loss": 0.0381, "grad_norm": 15.118651390075684, "learning_rate": 9.490909090909092e-06, "num_tokens": 324945.0, "completions/mean_length": 51.5, "completions/min_length": 38.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 51.5, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.8344469666481018, "rewards/meter/std": 0.3321688175201416, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9956766963005066, "rewards/repeat_soft/std": 0.00431094691157341, "rewards/judge_quality/mean": 0.5437500476837158, "rewards/judge_quality/std": 0.274170845746994, "rewards/total_composite/mean": 0.7881938219070435, "rewards/total_composite/std": 0.18166033923625946, "reward": 0.7881938219070435, "reward_std": 0.18166033923625946, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20305199921131134, "sampling/sampling_logp_difference/max": 1.4444313049316406, "sampling/importance_sampling_ratio/min": 0.23588019609451294, "sampling/importance_sampling_ratio/mean": 1.033147931098938, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.972639039158821, "clip_ratio/low_mean": 0.08102983422577381, "clip_ratio/low_min": 0.08102983422577381, "clip_ratio/high_mean": 0.10669679567217827, "clip_ratio/high_max": 0.10669679567217827, "clip_ratio/region_mean": 0.18772662989795208, "reward_total_mean": 0.7881938219070435, "reward_meter_mean": 0.8344469666481018, "reward_meter_std": 0.3321688175201416, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9956766963005066, "reward_repeat_soft_std": 0.00431094691157341, "reward_judge_quality_mean": 0.5437500476837158, "reward_judge_quality_std": 0.274170845746994, "reward_total_composite_mean": 0.7881938219070435, "reward_total_composite_std": 0.18166033923625946} +{"timestamp_utc": "2026-04-12T22:38:10Z", "mode": "train", "global_step": 170, "epoch": 0.017076845806127575, "loss": -0.1136, "grad_norm": 11.33247184753418, "learning_rate": 9.487878787878788e-06, "num_tokens": 326605.0, "completions/mean_length": 60.5, "completions/min_length": 32.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 60.5, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.6450315713882446, "rewards/meter/std": 0.43944719433784485, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.994788408279419, "rewards/repeat_soft/std": 0.006934575270861387, "rewards/judge_quality/mean": 0.5062500238418579, "rewards/judge_quality/std": 0.21084439754486084, "rewards/total_composite/mean": 0.691618025302887, "rewards/total_composite/std": 0.21619349718093872, "reward": 0.691618025302887, "reward_std": 0.21619349718093872, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19176585972309113, "sampling/sampling_logp_difference/max": 1.2366342544555664, "sampling/importance_sampling_ratio/min": 0.29035985469818115, "sampling/importance_sampling_ratio/mean": 1.0253686904907227, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2508623600006104, "clip_ratio/low_mean": 0.06805959902703762, "clip_ratio/low_min": 0.06805959902703762, "clip_ratio/high_mean": 0.10586905479431152, "clip_ratio/high_max": 0.10586905479431152, "clip_ratio/region_mean": 0.17392865382134914, "reward_total_mean": 0.691618025302887, "reward_meter_mean": 0.6450315713882446, "reward_meter_std": 0.43944719433784485, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.994788408279419, "reward_repeat_soft_std": 0.006934575270861387, "reward_judge_quality_mean": 0.5062500238418579, "reward_judge_quality_std": 0.21084439754486084, "reward_total_composite_mean": 0.691618025302887, "reward_total_composite_std": 0.21619349718093872} +{"timestamp_utc": "2026-04-12T22:38:17Z", "mode": "train", "global_step": 171, "epoch": 0.017177297840281266, "loss": 0.0733, "grad_norm": 12.215380668640137, "learning_rate": 9.484848484848485e-06, "num_tokens": 328431.0, "completions/mean_length": 62.25, "completions/min_length": 56.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.25, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.7637922167778015, "rewards/meter/std": 0.3509538173675537, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9932276606559753, "rewards/repeat_soft/std": 0.01168556697666645, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.7212792634963989, "rewards/total_composite/std": 0.15578024089336395, "reward": 0.7212792634963989, "reward_std": 0.15578024089336395, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18471170961856842, "sampling/sampling_logp_difference/max": 2.0597496032714844, "sampling/importance_sampling_ratio/min": 0.12748588621616364, "sampling/importance_sampling_ratio/mean": 1.0541846752166748, "sampling/importance_sampling_ratio/max": 1.9332194328308105, "entropy": 1.9895541220903397, "clip_ratio/low_mean": 0.07853026315569878, "clip_ratio/low_min": 0.07853026315569878, "clip_ratio/high_mean": 0.11501851491630077, "clip_ratio/high_max": 0.11501851491630077, "clip_ratio/region_mean": 0.19354877807199955, "reward_total_mean": 0.7212792634963989, "reward_meter_mean": 0.7637922167778015, "reward_meter_std": 0.3509538173675537, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9932276606559753, "reward_repeat_soft_std": 0.01168556697666645, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.7212792634963989, "reward_total_composite_std": 0.15578024089336395} +{"timestamp_utc": "2026-04-12T22:38:27Z", "mode": "train", "global_step": 172, "epoch": 0.017277749874434957, "loss": 0.0415, "grad_norm": 13.837084770202637, "learning_rate": 9.481818181818182e-06, "num_tokens": 330178.0, "completions/mean_length": 55.375, "completions/min_length": 50.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.375, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.805716872215271, "rewards/meter/std": 0.3166608512401581, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9853614568710327, "rewards/repeat_soft/std": 0.01756989397108555, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.7206087112426758, "rewards/total_composite/std": 0.1358690857887268, "reward": 0.7206087112426758, "reward_std": 0.1358690708875656, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13992270827293396, "sampling/sampling_logp_difference/max": 1.246485710144043, "sampling/importance_sampling_ratio/min": 0.2875134348869324, "sampling/importance_sampling_ratio/mean": 0.9956533908843994, "sampling/importance_sampling_ratio/max": 1.9486217498779297, "entropy": 1.2204567715525627, "clip_ratio/low_mean": 0.012500000186264515, "clip_ratio/low_min": 0.012500000186264515, "clip_ratio/high_mean": 0.1423769574612379, "clip_ratio/high_max": 0.1423769574612379, "clip_ratio/region_mean": 0.15487695764750242, "reward_total_mean": 0.7206087112426758, "reward_meter_mean": 0.805716872215271, "reward_meter_std": 0.3166608512401581, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9853614568710327, "reward_repeat_soft_std": 0.01756989397108555, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.7206087112426758, "reward_total_composite_std": 0.1358690857887268} +{"timestamp_utc": "2026-04-12T22:38:34Z", "mode": "train", "global_step": 173, "epoch": 0.017378201908588648, "loss": 0.0253, "grad_norm": 17.61992645263672, "learning_rate": 9.47878787878788e-06, "num_tokens": 331758.0, "completions/mean_length": 33.5, "completions/min_length": 26.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.5, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.46493038535118103, "rewards/meter/std": 0.39277565479278564, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.6650000214576721, "rewards/judge_quality/std": 0.3616628050804138, "rewards/total_composite/mean": 0.6549686789512634, "rewards/total_composite/std": 0.2746511399745941, "reward": 0.6549686789512634, "reward_std": 0.2746511399745941, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22067639231681824, "sampling/sampling_logp_difference/max": 1.2074518203735352, "sampling/importance_sampling_ratio/min": 0.29895809292793274, "sampling/importance_sampling_ratio/mean": 0.9965046644210815, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8067358434200287, "clip_ratio/low_mean": 0.09746044129133224, "clip_ratio/low_min": 0.09746044129133224, "clip_ratio/high_mean": 0.15546264499425888, "clip_ratio/high_max": 0.15546264499425888, "clip_ratio/region_mean": 0.2529230862855911, "reward_total_mean": 0.6549686789512634, "reward_meter_mean": 0.46493038535118103, "reward_meter_std": 0.39277565479278564, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.6650000214576721, "reward_judge_quality_std": 0.3616628050804138, "reward_total_composite_mean": 0.6549686789512634, "reward_total_composite_std": 0.2746511399745941} +{"timestamp_utc": "2026-04-12T22:38:41Z", "mode": "train", "global_step": 174, "epoch": 0.01747865394274234, "loss": 0.0398, "grad_norm": 11.458572387695312, "learning_rate": 9.475757575757577e-06, "num_tokens": 333544.0, "completions/mean_length": 66.25, "completions/min_length": 57.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.25, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.7040383815765381, "rewards/meter/std": 0.4332011938095093, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9906487464904785, "rewards/repeat_soft/std": 0.0084311468526721, "rewards/judge_quality/mean": 0.3349999785423279, "rewards/judge_quality/std": 0.09086881577968597, "rewards/total_composite/mean": 0.5697088241577148, "rewards/total_composite/std": 0.3074597716331482, "reward": 0.5697088241577148, "reward_std": 0.3074597418308258, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21156510710716248, "sampling/sampling_logp_difference/max": 1.6912975311279297, "sampling/importance_sampling_ratio/min": 0.18428026139736176, "sampling/importance_sampling_ratio/mean": 1.035260558128357, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.488626852631569, "clip_ratio/low_mean": 0.062451123259961605, "clip_ratio/low_min": 0.062451123259961605, "clip_ratio/high_mean": 0.11698358226567507, "clip_ratio/high_max": 0.11698358226567507, "clip_ratio/region_mean": 0.17943470552563667, "reward_total_mean": 0.5697088241577148, "reward_meter_mean": 0.7040383815765381, "reward_meter_std": 0.4332011938095093, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9906487464904785, "reward_repeat_soft_std": 0.0084311468526721, "reward_judge_quality_mean": 0.3349999785423279, "reward_judge_quality_std": 0.09086881577968597, "reward_total_composite_mean": 0.5697088241577148, "reward_total_composite_std": 0.3074597716331482} +{"timestamp_utc": "2026-04-12T22:38:48Z", "mode": "train", "global_step": 175, "epoch": 0.01757910597689603, "loss": 0.011, "grad_norm": 12.192331314086914, "learning_rate": 9.472727272727274e-06, "num_tokens": 335228.0, "completions/mean_length": 55.5, "completions/min_length": 39.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 55.5, "completions/min_terminated_length": 39.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.6966537833213806, "rewards/meter/std": 0.3916904330253601, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9995148181915283, "rewards/repeat_soft/std": 0.001061148359440267, "rewards/judge_quality/mean": 0.5612499713897705, "rewards/judge_quality/std": 0.25614938139915466, "rewards/total_composite/mean": 0.7318206429481506, "rewards/total_composite/std": 0.21793721616268158, "reward": 0.7318206429481506, "reward_std": 0.2179371863603592, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22025910019874573, "sampling/sampling_logp_difference/max": 1.8662443161010742, "sampling/importance_sampling_ratio/min": 0.15470360219478607, "sampling/importance_sampling_ratio/mean": 1.003103256225586, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9368027225136757, "clip_ratio/low_mean": 0.06872858293354511, "clip_ratio/low_min": 0.06872858293354511, "clip_ratio/high_mean": 0.1247611390426755, "clip_ratio/high_max": 0.1247611390426755, "clip_ratio/region_mean": 0.1934897219762206, "reward_total_mean": 0.7318206429481506, "reward_meter_mean": 0.6966537833213806, "reward_meter_std": 0.3916904330253601, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9995148181915283, "reward_repeat_soft_std": 0.001061148359440267, "reward_judge_quality_mean": 0.5612499713897705, "reward_judge_quality_std": 0.25614938139915466, "reward_total_composite_mean": 0.7318206429481506, "reward_total_composite_std": 0.21793721616268158} +{"timestamp_utc": "2026-04-12T22:38:55Z", "mode": "train", "global_step": 176, "epoch": 0.017679558011049725, "loss": 0.0106, "grad_norm": 17.778940200805664, "learning_rate": 9.469696969696971e-06, "num_tokens": 336807.0, "completions/mean_length": 39.375, "completions/min_length": 34.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.375, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.636339008808136, "rewards/meter/std": 0.36255964636802673, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9961650967597961, "rewards/repeat_soft/std": 0.004527193494141102, "rewards/judge_quality/mean": 0.5487500429153442, "rewards/judge_quality/std": 0.22937415540218353, "rewards/total_composite/mean": 0.6502262353897095, "rewards/total_composite/std": 0.27786290645599365, "reward": 0.6502262353897095, "reward_std": 0.27786290645599365, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2240709811449051, "sampling/sampling_logp_difference/max": 2.3256683349609375, "sampling/importance_sampling_ratio/min": 0.09771811217069626, "sampling/importance_sampling_ratio/mean": 1.0258606672286987, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.152998335659504, "clip_ratio/low_mean": 0.049544818699359894, "clip_ratio/low_min": 0.049544818699359894, "clip_ratio/high_mean": 0.14569118106737733, "clip_ratio/high_max": 0.14569118106737733, "clip_ratio/region_mean": 0.19523599976673722, "reward_total_mean": 0.6502262353897095, "reward_meter_mean": 0.636339008808136, "reward_meter_std": 0.36255964636802673, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9961650967597961, "reward_repeat_soft_std": 0.004527193494141102, "reward_judge_quality_mean": 0.5487500429153442, "reward_judge_quality_std": 0.22937415540218353, "reward_total_composite_mean": 0.6502262353897095, "reward_total_composite_std": 0.27786290645599365} +{"timestamp_utc": "2026-04-12T22:39:01Z", "mode": "train", "global_step": 177, "epoch": 0.017780010045203416, "loss": 0.1156, "grad_norm": 13.959003448486328, "learning_rate": 9.466666666666667e-06, "num_tokens": 338529.0, "completions/mean_length": 64.25, "completions/min_length": 56.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.25, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.6710958480834961, "rewards/meter/std": 0.39831477403640747, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9918336868286133, "rewards/repeat_soft/std": 0.021483546122908592, "rewards/judge_quality/mean": 0.44875001907348633, "rewards/judge_quality/std": 0.21256513893604279, "rewards/total_composite/mean": 0.6858015060424805, "rewards/total_composite/std": 0.2134513258934021, "reward": 0.6858015060424805, "reward_std": 0.2134513109922409, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1920711249113083, "sampling/sampling_logp_difference/max": 1.3209571838378906, "sampling/importance_sampling_ratio/min": 0.26687970757484436, "sampling/importance_sampling_ratio/mean": 1.0333881378173828, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0870207250118256, "clip_ratio/low_mean": 0.048370727337896824, "clip_ratio/low_min": 0.048370727337896824, "clip_ratio/high_mean": 0.10638105683028698, "clip_ratio/high_max": 0.10638105683028698, "clip_ratio/region_mean": 0.1547517841681838, "reward_total_mean": 0.6858015060424805, "reward_meter_mean": 0.6710958480834961, "reward_meter_std": 0.39831477403640747, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9918336868286133, "reward_repeat_soft_std": 0.021483546122908592, "reward_judge_quality_mean": 0.44875001907348633, "reward_judge_quality_std": 0.21256513893604279, "reward_total_composite_mean": 0.6858015060424805, "reward_total_composite_std": 0.2134513258934021} +{"timestamp_utc": "2026-04-12T22:39:11Z", "mode": "train", "global_step": 178, "epoch": 0.017880462079357107, "loss": 0.0151, "grad_norm": 6.557941913604736, "learning_rate": 9.463636363636364e-06, "num_tokens": 341373.0, "completions/mean_length": 157.5, "completions/min_length": 128.0, "completions/max_length": 186.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 157.5, "completions/min_terminated_length": 128.0, "completions/max_terminated_length": 186.0, "rewards/meter/mean": 0.8198948502540588, "rewards/meter/std": 0.29119426012039185, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9922699332237244, "rewards/repeat_soft/std": 0.010870981961488724, "rewards/judge_quality/mean": 0.34375, "rewards/judge_quality/std": 0.10966669768095016, "rewards/total_composite/mean": 0.7213046550750732, "rewards/total_composite/std": 0.15226486325263977, "reward": 0.7213046550750732, "reward_std": 0.15226484835147858, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18987901508808136, "sampling/sampling_logp_difference/max": 1.669508934020996, "sampling/importance_sampling_ratio/min": 0.18833953142166138, "sampling/importance_sampling_ratio/mean": 1.0420938730239868, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.5444493293762207, "clip_ratio/low_mean": 0.03958333469927311, "clip_ratio/low_min": 0.03958333469927311, "clip_ratio/high_mean": 0.1338949054479599, "clip_ratio/high_max": 0.1338949054479599, "clip_ratio/region_mean": 0.173478240147233, "reward_total_mean": 0.7213046550750732, "reward_meter_mean": 0.8198948502540588, "reward_meter_std": 0.29119426012039185, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9922699332237244, "reward_repeat_soft_std": 0.010870981961488724, "reward_judge_quality_mean": 0.34375, "reward_judge_quality_std": 0.10966669768095016, "reward_total_composite_mean": 0.7213046550750732, "reward_total_composite_std": 0.15226486325263977} +{"timestamp_utc": "2026-04-12T22:39:18Z", "mode": "train", "global_step": 179, "epoch": 0.0179809141135108, "loss": 0.0451, "grad_norm": 7.193534851074219, "learning_rate": 9.460606060606061e-06, "num_tokens": 343837.0, "completions/mean_length": 132.0, "completions/min_length": 120.0, "completions/max_length": 147.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 132.0, "completions/min_terminated_length": 120.0, "completions/max_terminated_length": 147.0, "rewards/meter/mean": 0.6529181599617004, "rewards/meter/std": 0.33133381605148315, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9972844123840332, "rewards/repeat_soft/std": 0.00252185994759202, "rewards/judge_quality/mean": 0.4362499713897705, "rewards/judge_quality/std": 0.12916629016399384, "rewards/total_composite/mean": 0.6744166016578674, "rewards/total_composite/std": 0.14660386741161346, "reward": 0.6744166016578674, "reward_std": 0.14660383760929108, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18584877252578735, "sampling/sampling_logp_difference/max": 1.4600629806518555, "sampling/importance_sampling_ratio/min": 0.23222164809703827, "sampling/importance_sampling_ratio/mean": 1.0315757989883423, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.474087432026863, "clip_ratio/low_mean": 0.06770833395421505, "clip_ratio/low_min": 0.06770833395421505, "clip_ratio/high_mean": 0.11158846318721771, "clip_ratio/high_max": 0.11158846318721771, "clip_ratio/region_mean": 0.17929679714143276, "reward_total_mean": 0.6744166016578674, "reward_meter_mean": 0.6529181599617004, "reward_meter_std": 0.33133381605148315, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9972844123840332, "reward_repeat_soft_std": 0.00252185994759202, "reward_judge_quality_mean": 0.4362499713897705, "reward_judge_quality_std": 0.12916629016399384, "reward_total_composite_mean": 0.6744166016578674, "reward_total_composite_std": 0.14660386741161346} +{"timestamp_utc": "2026-04-12T22:39:26Z", "mode": "train", "global_step": 180, "epoch": 0.01808136614766449, "loss": -0.0355, "grad_norm": 10.06341552734375, "learning_rate": 9.457575757575759e-06, "num_tokens": 346293.0, "completions/mean_length": 109.0, "completions/min_length": 88.0, "completions/max_length": 132.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 109.0, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 132.0, "rewards/meter/mean": 0.4708794951438904, "rewards/meter/std": 0.40037837624549866, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.13363061845302582, "rewards/hard_gate/mean": 0.875, "rewards/hard_gate/std": 0.3535533845424652, "rewards/repeat_soft/mean": 0.9943361282348633, "rewards/repeat_soft/std": 0.00587871577590704, "rewards/judge_quality/mean": 0.3100000023841858, "rewards/judge_quality/std": 0.12351980060338974, "rewards/total_composite/mean": 0.45544177293777466, "rewards/total_composite/std": 0.25155875086784363, "reward": 0.45544177293777466, "reward_std": 0.25155875086784363, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20670326054096222, "sampling/sampling_logp_difference/max": 1.546766757965088, "sampling/importance_sampling_ratio/min": 0.21293532848358154, "sampling/importance_sampling_ratio/mean": 1.0337767601013184, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.5227967500686646, "clip_ratio/low_mean": 0.1189667284488678, "clip_ratio/low_min": 0.1189667284488678, "clip_ratio/high_mean": 0.0869832169264555, "clip_ratio/high_max": 0.0869832169264555, "clip_ratio/region_mean": 0.2059499453753233, "reward_total_mean": 0.45544177293777466, "reward_meter_mean": 0.4708794951438904, "reward_meter_std": 0.40037837624549866, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.13363061845302582, "reward_hard_gate_mean": 0.875, "reward_hard_gate_std": 0.3535533845424652, "reward_repeat_soft_mean": 0.9943361282348633, "reward_repeat_soft_std": 0.00587871577590704, "reward_judge_quality_mean": 0.3100000023841858, "reward_judge_quality_std": 0.12351980060338974, "reward_total_composite_mean": 0.45544177293777466, "reward_total_composite_std": 0.25155875086784363} +{"timestamp_utc": "2026-04-12T22:39:32Z", "mode": "train", "global_step": 181, "epoch": 0.01818181818181818, "loss": 0.0199, "grad_norm": 11.772783279418945, "learning_rate": 9.454545454545456e-06, "num_tokens": 348100.0, "completions/mean_length": 65.875, "completions/min_length": 59.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.875, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.7274728417396545, "rewards/meter/std": 0.3568710684776306, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9997270107269287, "rewards/repeat_soft/std": 0.0007720249122940004, "rewards/judge_quality/mean": 0.5212500095367432, "rewards/judge_quality/std": 0.25837060809135437, "rewards/total_composite/mean": 0.733710527420044, "rewards/total_composite/std": 0.15925608575344086, "reward": 0.733710527420044, "reward_std": 0.15925607085227966, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16465792059898376, "sampling/sampling_logp_difference/max": 0.9982872009277344, "sampling/importance_sampling_ratio/min": 0.36851009726524353, "sampling/importance_sampling_ratio/mean": 1.0380871295928955, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.9717163443565369, "clip_ratio/low_mean": 0.04475616663694382, "clip_ratio/low_min": 0.04475616663694382, "clip_ratio/high_mean": 0.09800355415791273, "clip_ratio/high_max": 0.09800355415791273, "clip_ratio/region_mean": 0.14275972079485655, "reward_total_mean": 0.733710527420044, "reward_meter_mean": 0.7274728417396545, "reward_meter_std": 0.3568710684776306, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9997270107269287, "reward_repeat_soft_std": 0.0007720249122940004, "reward_judge_quality_mean": 0.5212500095367432, "reward_judge_quality_std": 0.25837060809135437, "reward_total_composite_mean": 0.733710527420044, "reward_total_composite_std": 0.15925608575344086} +{"timestamp_utc": "2026-04-12T22:39:38Z", "mode": "train", "global_step": 182, "epoch": 0.018282270215971872, "loss": 0.1234, "grad_norm": 28.719383239746094, "learning_rate": 9.451515151515153e-06, "num_tokens": 349735.0, "completions/mean_length": 40.375, "completions/min_length": 26.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.375, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.6474294662475586, "rewards/meter/std": 0.36743947863578796, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9840507507324219, "rewards/repeat_soft/std": 0.022401679307222366, "rewards/judge_quality/mean": 0.5149999856948853, "rewards/judge_quality/std": 0.2676885426044464, "rewards/total_composite/mean": 0.6942483186721802, "rewards/total_composite/std": 0.19757404923439026, "reward": 0.6942483186721802, "reward_std": 0.19757404923439026, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2344406545162201, "sampling/sampling_logp_difference/max": 2.6010918617248535, "sampling/importance_sampling_ratio/min": 0.07419252395629883, "sampling/importance_sampling_ratio/mean": 0.9845522046089172, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.226590447127819, "clip_ratio/low_mean": 0.03371212258934975, "clip_ratio/low_min": 0.03371212258934975, "clip_ratio/high_mean": 0.17729218304157257, "clip_ratio/high_max": 0.17729218304157257, "clip_ratio/region_mean": 0.21100430563092232, "reward_total_mean": 0.6942483186721802, "reward_meter_mean": 0.6474294662475586, "reward_meter_std": 0.36743947863578796, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9840507507324219, "reward_repeat_soft_std": 0.022401679307222366, "reward_judge_quality_mean": 0.5149999856948853, "reward_judge_quality_std": 0.2676885426044464, "reward_total_composite_mean": 0.6942483186721802, "reward_total_composite_std": 0.19757404923439026} +{"timestamp_utc": "2026-04-12T22:39:45Z", "mode": "train", "global_step": 183, "epoch": 0.018382722250125567, "loss": 0.1097, "grad_norm": 15.859298706054688, "learning_rate": 9.448484848484849e-06, "num_tokens": 351498.0, "completions/mean_length": 47.375, "completions/min_length": 40.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.375, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.2945380210876465, "rewards/meter/std": 0.2874630093574524, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9951648116111755, "rewards/repeat_soft/std": 0.005753299221396446, "rewards/judge_quality/mean": 0.4975000023841858, "rewards/judge_quality/std": 0.17136012017726898, "rewards/total_composite/mean": 0.5313085913658142, "rewards/total_composite/std": 0.13464023172855377, "reward": 0.5313085913658142, "reward_std": 0.13464021682739258, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22787530720233917, "sampling/sampling_logp_difference/max": 2.0343985557556152, "sampling/importance_sampling_ratio/min": 0.1307591050863266, "sampling/importance_sampling_ratio/mean": 1.0067754983901978, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.418655201792717, "clip_ratio/low_mean": 0.12804551888257265, "clip_ratio/low_min": 0.12804551888257265, "clip_ratio/high_mean": 0.09737318754196167, "clip_ratio/high_max": 0.09737318754196167, "clip_ratio/region_mean": 0.22541870642453432, "reward_total_mean": 0.5313085913658142, "reward_meter_mean": 0.2945380210876465, "reward_meter_std": 0.2874630093574524, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9951648116111755, "reward_repeat_soft_std": 0.005753299221396446, "reward_judge_quality_mean": 0.4975000023841858, "reward_judge_quality_std": 0.17136012017726898, "reward_total_composite_mean": 0.5313085913658142, "reward_total_composite_std": 0.13464023172855377} +{"timestamp_utc": "2026-04-12T22:39:52Z", "mode": "train", "global_step": 184, "epoch": 0.018483174284279258, "loss": 0.0429, "grad_norm": 10.949633598327637, "learning_rate": 9.445454545454546e-06, "num_tokens": 353266.0, "completions/mean_length": 65.0, "completions/min_length": 59.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.0, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.6457832455635071, "rewards/meter/std": 0.40137815475463867, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9987822771072388, "rewards/repeat_soft/std": 0.001760914921760559, "rewards/judge_quality/mean": 0.4099999964237213, "rewards/judge_quality/std": 0.06633248925209045, "rewards/total_composite/mean": 0.6541056632995605, "rewards/total_composite/std": 0.18264658749103546, "reward": 0.6541056632995605, "reward_std": 0.18264657258987427, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17986838519573212, "sampling/sampling_logp_difference/max": 1.1473078727722168, "sampling/importance_sampling_ratio/min": 0.3174903392791748, "sampling/importance_sampling_ratio/mean": 1.0141305923461914, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1695086508989334, "clip_ratio/low_mean": 0.055078125558793545, "clip_ratio/low_min": 0.055078125558793545, "clip_ratio/high_mean": 0.12554166465997696, "clip_ratio/high_max": 0.12554166465997696, "clip_ratio/region_mean": 0.1806197902187705, "reward_total_mean": 0.6541056632995605, "reward_meter_mean": 0.6457832455635071, "reward_meter_std": 0.40137815475463867, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9987822771072388, "reward_repeat_soft_std": 0.001760914921760559, "reward_judge_quality_mean": 0.4099999964237213, "reward_judge_quality_std": 0.06633248925209045, "reward_total_composite_mean": 0.6541056632995605, "reward_total_composite_std": 0.18264658749103546} +{"timestamp_utc": "2026-04-12T22:39:58Z", "mode": "train", "global_step": 185, "epoch": 0.01858362631843295, "loss": 0.0067, "grad_norm": 18.104721069335938, "learning_rate": 9.442424242424243e-06, "num_tokens": 354728.0, "completions/mean_length": 30.75, "completions/min_length": 26.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.75, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.6661310791969299, "rewards/meter/std": 0.34061819314956665, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9516666531562805, "rewards/repeat_soft/std": 0.024608036503195763, "rewards/judge_quality/mean": 0.35999998450279236, "rewards/judge_quality/std": 0.09165150672197342, "rewards/total_composite/mean": 0.6529256105422974, "rewards/total_composite/std": 0.15953470766544342, "reward": 0.6529256105422974, "reward_std": 0.15953470766544342, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17186814546585083, "sampling/sampling_logp_difference/max": 1.139305591583252, "sampling/importance_sampling_ratio/min": 0.3200412094593048, "sampling/importance_sampling_ratio/mean": 1.0297634601593018, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8394776284694672, "clip_ratio/low_mean": 0.04160380829125643, "clip_ratio/low_min": 0.04160380829125643, "clip_ratio/high_mean": 0.10947296489030123, "clip_ratio/high_max": 0.10947296489030123, "clip_ratio/region_mean": 0.15107677318155766, "reward_total_mean": 0.6529256105422974, "reward_meter_mean": 0.6661310791969299, "reward_meter_std": 0.34061819314956665, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9516666531562805, "reward_repeat_soft_std": 0.024608036503195763, "reward_judge_quality_mean": 0.35999998450279236, "reward_judge_quality_std": 0.09165150672197342, "reward_total_composite_mean": 0.6529256105422974, "reward_total_composite_std": 0.15953470766544342} +{"timestamp_utc": "2026-04-12T22:40:06Z", "mode": "train", "global_step": 186, "epoch": 0.01868407835258664, "loss": 0.0512, "grad_norm": 11.477437019348145, "learning_rate": 9.43939393939394e-06, "num_tokens": 356878.0, "completions/mean_length": 85.75, "completions/min_length": 81.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 85.75, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.33967459201812744, "rewards/meter/std": 0.3127252459526062, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9931557774543762, "rewards/repeat_soft/std": 0.00534589309245348, "rewards/judge_quality/mean": 0.3987500071525574, "rewards/judge_quality/std": 0.06010407209396362, "rewards/total_composite/mean": 0.5217941403388977, "rewards/total_composite/std": 0.14443467557430267, "reward": 0.5217941403388977, "reward_std": 0.14443467557430267, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21810176968574524, "sampling/sampling_logp_difference/max": 4.856741428375244, "sampling/importance_sampling_ratio/min": 0.007775780279189348, "sampling/importance_sampling_ratio/mean": 1.0244901180267334, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3468519300222397, "clip_ratio/low_mean": 0.11296561546623707, "clip_ratio/low_min": 0.11296561546623707, "clip_ratio/high_mean": 0.07902637496590614, "clip_ratio/high_max": 0.07902637496590614, "clip_ratio/region_mean": 0.1919919904321432, "reward_total_mean": 0.5217941403388977, "reward_meter_mean": 0.33967459201812744, "reward_meter_std": 0.3127252459526062, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9931557774543762, "reward_repeat_soft_std": 0.00534589309245348, "reward_judge_quality_mean": 0.3987500071525574, "reward_judge_quality_std": 0.06010407209396362, "reward_total_composite_mean": 0.5217941403388977, "reward_total_composite_std": 0.14443467557430267} +{"timestamp_utc": "2026-04-12T22:40:15Z", "mode": "train", "global_step": 187, "epoch": 0.01878453038674033, "loss": -0.0318, "grad_norm": 5.909311294555664, "learning_rate": 9.436363636363636e-06, "num_tokens": 360033.0, "completions/mean_length": 196.375, "completions/min_length": 166.0, "completions/max_length": 226.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 196.375, "completions/min_terminated_length": 166.0, "completions/max_terminated_length": 226.0, "rewards/meter/mean": 0.7298188805580139, "rewards/meter/std": 0.3511499762535095, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/hard_gate/mean": 0.75, "rewards/hard_gate/std": 0.4629100561141968, "rewards/repeat_soft/mean": 0.9943335056304932, "rewards/repeat_soft/std": 0.00467671500518918, "rewards/judge_quality/mean": 0.29249998927116394, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.5305139422416687, "rewards/total_composite/std": 0.3508707284927368, "reward": 0.5305139422416687, "reward_std": 0.35087069869041443, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1829221248626709, "sampling/sampling_logp_difference/max": 1.5122761726379395, "sampling/importance_sampling_ratio/min": 0.22040772438049316, "sampling/importance_sampling_ratio/mean": 1.0364093780517578, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3696429282426834, "clip_ratio/low_mean": 0.05853912979364395, "clip_ratio/low_min": 0.05853912979364395, "clip_ratio/high_mean": 0.10988106019794941, "clip_ratio/high_max": 0.10988106019794941, "clip_ratio/region_mean": 0.16842018999159336, "reward_total_mean": 0.5305139422416687, "reward_meter_mean": 0.7298188805580139, "reward_meter_std": 0.3511499762535095, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_hard_gate_mean": 0.75, "reward_hard_gate_std": 0.4629100561141968, "reward_repeat_soft_mean": 0.9943335056304932, "reward_repeat_soft_std": 0.00467671500518918, "reward_judge_quality_mean": 0.29249998927116394, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.5305139422416687, "reward_total_composite_std": 0.3508707284927368} +{"timestamp_utc": "2026-04-12T22:40:22Z", "mode": "train", "global_step": 188, "epoch": 0.018884982420894023, "loss": -0.0028, "grad_norm": 10.292993545532227, "learning_rate": 9.433333333333335e-06, "num_tokens": 361852.0, "completions/mean_length": 70.375, "completions/min_length": 66.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.375, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.9887566566467285, "rewards/meter/std": 0.022077880799770355, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9968456029891968, "rewards/repeat_soft/std": 0.0058901249431073666, "rewards/judge_quality/mean": 0.4099999964237213, "rewards/judge_quality/std": 0.06633248925209045, "rewards/total_composite/mean": 0.8176250457763672, "rewards/total_composite/std": 0.02950715832412243, "reward": 0.8176250457763672, "reward_std": 0.029507163912057877, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1563209444284439, "sampling/sampling_logp_difference/max": 1.2766375541687012, "sampling/importance_sampling_ratio/min": 0.2789737284183502, "sampling/importance_sampling_ratio/mean": 1.0275911092758179, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0365912318229675, "clip_ratio/low_mean": 0.016791045665740967, "clip_ratio/low_min": 0.016791045665740967, "clip_ratio/high_mean": 0.15039433352649212, "clip_ratio/high_max": 0.15039433352649212, "clip_ratio/region_mean": 0.16718537919223309, "reward_total_mean": 0.8176250457763672, "reward_meter_mean": 0.9887566566467285, "reward_meter_std": 0.022077880799770355, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9968456029891968, "reward_repeat_soft_std": 0.0058901249431073666, "reward_judge_quality_mean": 0.4099999964237213, "reward_judge_quality_std": 0.06633248925209045, "reward_total_composite_mean": 0.8176250457763672, "reward_total_composite_std": 0.02950715832412243} +{"timestamp_utc": "2026-04-12T22:40:29Z", "mode": "train", "global_step": 189, "epoch": 0.018985434455047714, "loss": 0.1136, "grad_norm": 26.591156005859375, "learning_rate": 9.43030303030303e-06, "num_tokens": 363424.0, "completions/mean_length": 40.5, "completions/min_length": 31.0, "completions/max_length": 46.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.5, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 46.0, "rewards/meter/mean": 0.3919089734554291, "rewards/meter/std": 0.3436703681945801, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9904083013534546, "rewards/repeat_soft/std": 0.012789538130164146, "rewards/judge_quality/mean": 0.7950000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.663899838924408, "rewards/total_composite/std": 0.15209518373012543, "reward": 0.663899838924408, "reward_std": 0.15209518373012543, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22401906549930573, "sampling/sampling_logp_difference/max": 1.9713678359985352, "sampling/importance_sampling_ratio/min": 0.1392662227153778, "sampling/importance_sampling_ratio/mean": 1.0248298645019531, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.4507956057786942, "clip_ratio/low_mean": 0.11950287409126759, "clip_ratio/low_min": 0.11950287409126759, "clip_ratio/high_mean": 0.10845978185534477, "clip_ratio/high_max": 0.10845978185534477, "clip_ratio/region_mean": 0.22796265594661236, "reward_total_mean": 0.663899838924408, "reward_meter_mean": 0.3919089734554291, "reward_meter_std": 0.3436703681945801, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9904083013534546, "reward_repeat_soft_std": 0.012789538130164146, "reward_judge_quality_mean": 0.7950000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.663899838924408, "reward_total_composite_std": 0.15209518373012543} +{"timestamp_utc": "2026-04-12T22:40:36Z", "mode": "train", "global_step": 190, "epoch": 0.019085886489201405, "loss": 0.1071, "grad_norm": 13.938616752624512, "learning_rate": 9.427272727272728e-06, "num_tokens": 365233.0, "completions/mean_length": 57.125, "completions/min_length": 49.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.125, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.7486714720726013, "rewards/meter/std": 0.38221657276153564, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9878576993942261, "rewards/repeat_soft/std": 0.017484119161963463, "rewards/judge_quality/mean": 0.4312500059604645, "rewards/judge_quality/std": 0.015526476316154003, "rewards/total_composite/mean": 0.7150629162788391, "rewards/total_composite/std": 0.17185485363006592, "reward": 0.7150629162788391, "reward_std": 0.17185483872890472, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1873229295015335, "sampling/sampling_logp_difference/max": 1.6582221984863281, "sampling/importance_sampling_ratio/min": 0.19047731161117554, "sampling/importance_sampling_ratio/mean": 1.0213521718978882, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.981810376048088, "clip_ratio/low_mean": 0.0448082871735096, "clip_ratio/low_min": 0.0448082871735096, "clip_ratio/high_mean": 0.16760712210088968, "clip_ratio/high_max": 0.16760712210088968, "clip_ratio/region_mean": 0.21241540927439928, "reward_total_mean": 0.7150629162788391, "reward_meter_mean": 0.7486714720726013, "reward_meter_std": 0.38221657276153564, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9878576993942261, "reward_repeat_soft_std": 0.017484119161963463, "reward_judge_quality_mean": 0.4312500059604645, "reward_judge_quality_std": 0.015526476316154003, "reward_total_composite_mean": 0.7150629162788391, "reward_total_composite_std": 0.17185485363006592} +{"timestamp_utc": "2026-04-12T22:40:44Z", "mode": "train", "global_step": 191, "epoch": 0.0191863385233551, "loss": 0.0827, "grad_norm": 6.0124125480651855, "learning_rate": 9.424242424242425e-06, "num_tokens": 368166.0, "completions/mean_length": 181.625, "completions/min_length": 158.0, "completions/max_length": 211.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 181.625, "completions/min_terminated_length": 158.0, "completions/max_terminated_length": 211.0, "rewards/meter/mean": 0.9360668659210205, "rewards/meter/std": 0.11463984847068787, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9874351620674133, "rewards/repeat_soft/std": 0.011888244189321995, "rewards/judge_quality/mean": 0.36500000953674316, "rewards/judge_quality/std": 0.10528871417045593, "rewards/total_composite/mean": 0.7757235765457153, "rewards/total_composite/std": 0.05351750925183296, "reward": 0.7757235765457153, "reward_std": 0.05351749435067177, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1810183823108673, "sampling/sampling_logp_difference/max": 1.7759523391723633, "sampling/importance_sampling_ratio/min": 0.16932211816310883, "sampling/importance_sampling_ratio/mean": 1.0242165327072144, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.2983481884002686, "clip_ratio/low_mean": 0.06891964096575975, "clip_ratio/low_min": 0.06891964096575975, "clip_ratio/high_mean": 0.0929587259888649, "clip_ratio/high_max": 0.0929587259888649, "clip_ratio/region_mean": 0.16187836695462465, "reward_total_mean": 0.7757235765457153, "reward_meter_mean": 0.9360668659210205, "reward_meter_std": 0.11463984847068787, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9874351620674133, "reward_repeat_soft_std": 0.011888244189321995, "reward_judge_quality_mean": 0.36500000953674316, "reward_judge_quality_std": 0.10528871417045593, "reward_total_composite_mean": 0.7757235765457153, "reward_total_composite_std": 0.05351750925183296} +{"timestamp_utc": "2026-04-12T22:40:50Z", "mode": "train", "global_step": 192, "epoch": 0.01928679055750879, "loss": 0.0511, "grad_norm": 20.492347717285156, "learning_rate": 9.421212121212122e-06, "num_tokens": 369877.0, "completions/mean_length": 42.875, "completions/min_length": 37.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.875, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.6187220215797424, "rewards/meter/std": 0.4059104025363922, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9923666715621948, "rewards/repeat_soft/std": 0.007227038033306599, "rewards/judge_quality/mean": 0.42750000953674316, "rewards/judge_quality/std": 0.013887302950024605, "rewards/total_composite/mean": 0.6559115648269653, "rewards/total_composite/std": 0.18223440647125244, "reward": 0.6559115648269653, "reward_std": 0.18223440647125244, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17871472239494324, "sampling/sampling_logp_difference/max": 1.7018980979919434, "sampling/importance_sampling_ratio/min": 0.18233710527420044, "sampling/importance_sampling_ratio/mean": 1.0145092010498047, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.3617488741874695, "clip_ratio/low_mean": 0.06380685605108738, "clip_ratio/low_min": 0.06380685605108738, "clip_ratio/high_mean": 0.12363456562161446, "clip_ratio/high_max": 0.12363456562161446, "clip_ratio/region_mean": 0.18744142167270184, "reward_total_mean": 0.6559115648269653, "reward_meter_mean": 0.6187220215797424, "reward_meter_std": 0.4059104025363922, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9923666715621948, "reward_repeat_soft_std": 0.007227038033306599, "reward_judge_quality_mean": 0.42750000953674316, "reward_judge_quality_std": 0.013887302950024605, "reward_total_composite_mean": 0.6559115648269653, "reward_total_composite_std": 0.18223440647125244} +{"timestamp_utc": "2026-04-12T22:40:57Z", "mode": "train", "global_step": 193, "epoch": 0.019387242591662482, "loss": 0.0257, "grad_norm": 8.760064125061035, "learning_rate": 9.418181818181818e-06, "num_tokens": 372067.0, "completions/mean_length": 95.75, "completions/min_length": 89.0, "completions/max_length": 102.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 95.75, "completions/min_terminated_length": 89.0, "completions/max_terminated_length": 102.0, "rewards/meter/mean": 0.7216939926147461, "rewards/meter/std": 0.38140353560447693, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9865915775299072, "rewards/repeat_soft/std": 0.015576460398733616, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.6802964806556702, "rewards/total_composite/std": 0.1582382768392563, "reward": 0.6802964806556702, "reward_std": 0.1582382768392563, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1715310662984848, "sampling/sampling_logp_difference/max": 1.0202453136444092, "sampling/importance_sampling_ratio/min": 0.37454289197921753, "sampling/importance_sampling_ratio/mean": 1.0401523113250732, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.0904775112867355, "clip_ratio/low_mean": 0.06285456381738186, "clip_ratio/low_min": 0.06285456381738186, "clip_ratio/high_mean": 0.10084935091435909, "clip_ratio/high_max": 0.10084935091435909, "clip_ratio/region_mean": 0.16370391473174095, "reward_total_mean": 0.6802964806556702, "reward_meter_mean": 0.7216939926147461, "reward_meter_std": 0.38140353560447693, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9865915775299072, "reward_repeat_soft_std": 0.015576460398733616, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.6802964806556702, "reward_total_composite_std": 0.1582382768392563} +{"timestamp_utc": "2026-04-12T22:41:03Z", "mode": "train", "global_step": 194, "epoch": 0.019487694625816173, "loss": 0.0545, "grad_norm": 9.997492790222168, "learning_rate": 9.415151515151515e-06, "num_tokens": 373859.0, "completions/mean_length": 67.0, "completions/min_length": 49.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9916137456893921, "rewards/meter/std": 0.006394678261131048, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.995553731918335, "rewards/repeat_soft/std": 0.0062927003018558025, "rewards/judge_quality/mean": 0.35624998807907104, "rewards/judge_quality/std": 0.08798335492610931, "rewards/total_composite/mean": 0.8026565313339233, "rewards/total_composite/std": 0.02471695840358734, "reward": 0.8026565313339233, "reward_std": 0.02471696026623249, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18926693499088287, "sampling/sampling_logp_difference/max": 1.1049785614013672, "sampling/importance_sampling_ratio/min": 0.33121800422668457, "sampling/importance_sampling_ratio/mean": 1.0358859300613403, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.3879302740097046, "clip_ratio/low_mean": 0.06041587330400944, "clip_ratio/low_min": 0.06041587330400944, "clip_ratio/high_mean": 0.12720229662954807, "clip_ratio/high_max": 0.12720229662954807, "clip_ratio/region_mean": 0.1876181699335575, "reward_total_mean": 0.8026565313339233, "reward_meter_mean": 0.9916137456893921, "reward_meter_std": 0.006394678261131048, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.995553731918335, "reward_repeat_soft_std": 0.0062927003018558025, "reward_judge_quality_mean": 0.35624998807907104, "reward_judge_quality_std": 0.08798335492610931, "reward_total_composite_mean": 0.8026565313339233, "reward_total_composite_std": 0.02471695840358734} +{"timestamp_utc": "2026-04-12T22:41:10Z", "mode": "train", "global_step": 195, "epoch": 0.019588146659969864, "loss": -0.0807, "grad_norm": 10.878082275390625, "learning_rate": 9.412121212121212e-06, "num_tokens": 375657.0, "completions/mean_length": 65.75, "completions/min_length": 38.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.75, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.5816943645477295, "rewards/meter/std": 0.37206828594207764, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9994540810585022, "rewards/repeat_soft/std": 0.0010108178248628974, "rewards/judge_quality/mean": 0.36000001430511475, "rewards/judge_quality/std": 0.09165150672197342, "rewards/total_composite/mean": 0.6103328466415405, "rewards/total_composite/std": 0.19548235833644867, "reward": 0.6103328466415405, "reward_std": 0.19548232853412628, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2016046941280365, "sampling/sampling_logp_difference/max": 1.774024486541748, "sampling/importance_sampling_ratio/min": 0.16964885592460632, "sampling/importance_sampling_ratio/mean": 1.0383213758468628, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.4231244772672653, "clip_ratio/low_mean": 0.08922833111137152, "clip_ratio/low_min": 0.08922833111137152, "clip_ratio/high_mean": 0.09207563661038876, "clip_ratio/high_max": 0.09207563661038876, "clip_ratio/region_mean": 0.18130396772176027, "reward_total_mean": 0.6103328466415405, "reward_meter_mean": 0.5816943645477295, "reward_meter_std": 0.37206828594207764, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9994540810585022, "reward_repeat_soft_std": 0.0010108178248628974, "reward_judge_quality_mean": 0.36000001430511475, "reward_judge_quality_std": 0.09165150672197342, "reward_total_composite_mean": 0.6103328466415405, "reward_total_composite_std": 0.19548235833644867} +{"timestamp_utc": "2026-04-12T22:41:16Z", "mode": "train", "global_step": 196, "epoch": 0.019688598694123555, "loss": 0.0618, "grad_norm": 17.475666046142578, "learning_rate": 9.40909090909091e-06, "num_tokens": 377070.0, "completions/mean_length": 27.625, "completions/min_length": 22.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 27.625, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.4827464818954468, "rewards/meter/std": 0.43062275648117065, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.53125, "rewards/judge_quality/std": 0.24833375215530396, "rewards/total_composite/mean": 0.6228609085083008, "rewards/total_composite/std": 0.22536708414554596, "reward": 0.6228609085083008, "reward_std": 0.22536706924438477, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17893603444099426, "sampling/sampling_logp_difference/max": 1.1922149658203125, "sampling/importance_sampling_ratio/min": 0.32511380314826965, "sampling/importance_sampling_ratio/mean": 1.019759178161621, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.8978001922369003, "clip_ratio/low_mean": 0.09404377825558186, "clip_ratio/low_min": 0.09404377825558186, "clip_ratio/high_mean": 0.11062751803547144, "clip_ratio/high_max": 0.11062751803547144, "clip_ratio/region_mean": 0.2046712962910533, "reward_total_mean": 0.6228609085083008, "reward_meter_mean": 0.4827464818954468, "reward_meter_std": 0.43062275648117065, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.53125, "reward_judge_quality_std": 0.24833375215530396, "reward_total_composite_mean": 0.6228609085083008, "reward_total_composite_std": 0.22536708414554596} +{"timestamp_utc": "2026-04-12T22:41:22Z", "mode": "train", "global_step": 197, "epoch": 0.019789050728277247, "loss": 0.1273, "grad_norm": 22.129459381103516, "learning_rate": 9.406060606060607e-06, "num_tokens": 378529.0, "completions/mean_length": 35.375, "completions/min_length": 30.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.375, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.8654494285583496, "rewards/meter/std": 0.34853827953338623, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9624999761581421, "rewards/repeat_soft/std": 0.0, "rewards/judge_quality/mean": 0.8025000095367432, "rewards/judge_quality/std": 0.22720351815223694, "rewards/total_composite/mean": 0.8764522671699524, "rewards/total_composite/std": 0.20609977841377258, "reward": 0.8764522671699524, "reward_std": 0.2060997486114502, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15852470695972443, "sampling/sampling_logp_difference/max": 0.9438915252685547, "sampling/importance_sampling_ratio/min": 0.3891106843948364, "sampling/importance_sampling_ratio/mean": 1.011351466178894, "sampling/importance_sampling_ratio/max": 1.956284761428833, "entropy": 1.4904959127306938, "clip_ratio/low_mean": 0.04118217155337334, "clip_ratio/low_min": 0.04118217155337334, "clip_ratio/high_mean": 0.12981951143592596, "clip_ratio/high_max": 0.12981951143592596, "clip_ratio/region_mean": 0.1710016829892993, "reward_total_mean": 0.8764522671699524, "reward_meter_mean": 0.8654494285583496, "reward_meter_std": 0.34853827953338623, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9624999761581421, "reward_repeat_soft_std": 0.0, "reward_judge_quality_mean": 0.8025000095367432, "reward_judge_quality_std": 0.22720351815223694, "reward_total_composite_mean": 0.8764522671699524, "reward_total_composite_std": 0.20609977841377258} +{"timestamp_utc": "2026-04-12T22:41:29Z", "mode": "train", "global_step": 198, "epoch": 0.019889502762430938, "loss": 0.1145, "grad_norm": 16.026708602905273, "learning_rate": 9.403030303030304e-06, "num_tokens": 380541.0, "completions/mean_length": 68.5, "completions/min_length": 61.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.5, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.4582224190235138, "rewards/meter/std": 0.3012732267379761, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9965095520019531, "rewards/repeat_soft/std": 0.002199555281549692, "rewards/judge_quality/mean": 0.5450000166893005, "rewards/judge_quality/std": 0.23145504295825958, "rewards/total_composite/mean": 0.6193510293960571, "rewards/total_composite/std": 0.18768593668937683, "reward": 0.6193510293960571, "reward_std": 0.18768592178821564, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.22508522868156433, "sampling/sampling_logp_difference/max": 1.8852663040161133, "sampling/importance_sampling_ratio/min": 0.1517886370420456, "sampling/importance_sampling_ratio/mean": 0.9763094186782837, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 1.054028607904911, "clip_ratio/low_mean": 0.13065637461841106, "clip_ratio/low_min": 0.13065637461841106, "clip_ratio/high_mean": 0.06116020958870649, "clip_ratio/high_max": 0.06116020958870649, "clip_ratio/region_mean": 0.19181658420711756, "reward_total_mean": 0.6193510293960571, "reward_meter_mean": 0.4582224190235138, "reward_meter_std": 0.3012732267379761, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9965095520019531, "reward_repeat_soft_std": 0.002199555281549692, "reward_judge_quality_mean": 0.5450000166893005, "reward_judge_quality_std": 0.23145504295825958, "reward_total_composite_mean": 0.6193510293960571, "reward_total_composite_std": 0.18768593668937683} +{"timestamp_utc": "2026-04-12T22:41:35Z", "mode": "train", "global_step": 199, "epoch": 0.019989954796584632, "loss": 0.0819, "grad_norm": 17.990541458129883, "learning_rate": 9.4e-06, "num_tokens": 381947.0, "completions/mean_length": 30.75, "completions/min_length": 26.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.75, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.9828565120697021, "rewards/meter/std": 0.01875847764313221, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9604166746139526, "rewards/repeat_soft/std": 0.00589255103841424, "rewards/judge_quality/mean": 0.36375001072883606, "rewards/judge_quality/std": 0.13265825808048248, "rewards/total_composite/mean": 0.7974521517753601, "rewards/total_composite/std": 0.037255607545375824, "reward": 0.7974521517753601, "reward_std": 0.037255603820085526, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18850040435791016, "sampling/sampling_logp_difference/max": 1.4986425638198853, "sampling/importance_sampling_ratio/min": 0.223433256149292, "sampling/importance_sampling_ratio/mean": 1.0070288181304932, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.056455612182617, "clip_ratio/low_mean": 0.04545454680919647, "clip_ratio/low_min": 0.04545454680919647, "clip_ratio/high_mean": 0.178089483641088, "clip_ratio/high_max": 0.178089483641088, "clip_ratio/region_mean": 0.22354403045028448, "reward_total_mean": 0.7974521517753601, "reward_meter_mean": 0.9828565120697021, "reward_meter_std": 0.01875847764313221, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9604166746139526, "reward_repeat_soft_std": 0.00589255103841424, "reward_judge_quality_mean": 0.36375001072883606, "reward_judge_quality_std": 0.13265825808048248, "reward_total_composite_mean": 0.7974521517753601, "reward_total_composite_std": 0.037255607545375824} +{"timestamp_utc": "2026-04-12T22:41:42Z", "mode": "train", "global_step": 200, "epoch": 0.020090406830738324, "loss": 0.0572, "grad_norm": 7.267741680145264, "learning_rate": 9.396969696969697e-06, "num_tokens": 384387.0, "completions/mean_length": 131.0, "completions/min_length": 111.0, "completions/max_length": 146.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 131.0, "completions/min_terminated_length": 111.0, "completions/max_terminated_length": 146.0, "rewards/meter/mean": 0.930094301700592, "rewards/meter/std": 0.12754081189632416, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.997650146484375, "rewards/repeat_soft/std": 0.0023139191325753927, "rewards/judge_quality/mean": 0.3774999976158142, "rewards/judge_quality/std": 0.07869470119476318, "rewards/total_composite/mean": 0.7815574407577515, "rewards/total_composite/std": 0.05706481263041496, "reward": 0.7815574407577515, "reward_std": 0.057064805179834366, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20789173245429993, "sampling/sampling_logp_difference/max": 2.3408408164978027, "sampling/importance_sampling_ratio/min": 0.09624667465686798, "sampling/importance_sampling_ratio/mean": 1.0487194061279297, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.758536249399185, "clip_ratio/low_mean": 0.05325818154960871, "clip_ratio/low_min": 0.05325818154960871, "clip_ratio/high_mean": 0.11834869347512722, "clip_ratio/high_max": 0.11834869347512722, "clip_ratio/region_mean": 0.17160687502473593, "reward_total_mean": 0.7815574407577515, "reward_meter_mean": 0.930094301700592, "reward_meter_std": 0.12754081189632416, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.997650146484375, "reward_repeat_soft_std": 0.0023139191325753927, "reward_judge_quality_mean": 0.3774999976158142, "reward_judge_quality_std": 0.07869470119476318, "reward_total_composite_mean": 0.7815574407577515, "reward_total_composite_std": 0.05706481263041496} +{"timestamp_utc": "2026-04-12T22:42:33Z", "mode": "eval", "global_step": 200, "epoch": 0.020090406830738324, "eval_loss": NaN, "eval_runtime": 51.3677, "eval_samples_per_second": 1.557, "eval_steps_per_second": 0.195, "eval_num_tokens": 384387.0, "eval_completions/mean_length": 98.4375, "eval_completions/min_length": 38.0, "eval_completions/max_length": 184.9, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 98.4375, "eval_completions/min_terminated_length": 38.0, "eval_completions/max_terminated_length": 184.9, "eval_rewards/meter/mean": 0.6527223944664001, "eval_rewards/meter/std": 0.3822064697742462, "eval_rewards/count_adherence/mean": 0.9860416531562806, "eval_rewards/count_adherence/std": 0.030216221511363984, "eval_rewards/hard_gate/mean": 0.975, "eval_rewards/hard_gate/std": 0.07071067690849304, "eval_rewards/repeat_soft/mean": 0.9887738764286041, "eval_rewards/repeat_soft/std": 0.015243505884427577, "eval_rewards/judge_quality/mean": 0.3777499973773956, "eval_rewards/judge_quality/std": 0.13215117231011392, "eval_rewards/total_composite/mean": 0.6343437016010285, "eval_rewards/total_composite/std": 0.2018197938799858, "eval_reward": 0.6343437016010285, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.1359984040260315, "eval_sampling/sampling_logp_difference/max": 1.2542716979980468, "eval_sampling/importance_sampling_ratio/min": 0.28756721019744874, "eval_sampling/importance_sampling_ratio/mean": 1.0338929533958434, "eval_sampling/importance_sampling_ratio/max": 1.5515777111053466, "eval_entropy": 2.141538119316101, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.6343437016010285, "eval_reward_meter_mean": 0.6527223944664001, "eval_reward_meter_std": 0.3822064697742462, "eval_reward_count_adherence_mean": 0.9860416531562806, "eval_reward_count_adherence_std": 0.030216221511363984, "eval_reward_hard_gate_mean": 0.975, "eval_reward_hard_gate_std": 0.07071067690849304, "eval_reward_repeat_soft_mean": 0.9887738764286041, "eval_reward_repeat_soft_std": 0.015243505884427577, "eval_reward_judge_quality_mean": 0.3777499973773956, "eval_reward_judge_quality_std": 0.13215117231011392, "eval_reward_total_composite_mean": 0.6343437016010285, "eval_reward_total_composite_std": 0.2018197938799858} +{"timestamp_utc": "2026-04-12T22:42:43Z", "mode": "train", "global_step": 201, "epoch": 0.020190858864892015, "loss": -0.0725, "grad_norm": 9.45182991027832, "learning_rate": 9.393939393939396e-06, "num_tokens": 386624.0, "completions/mean_length": 90.625, "completions/min_length": 53.0, "completions/max_length": 106.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 90.625, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.7137892246246338, "rewards/meter/std": 0.38705575466156006, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/hard_gate/mean": 1.0, "rewards/hard_gate/std": 0.0, "rewards/repeat_soft/mean": 0.9941529035568237, "rewards/repeat_soft/std": 0.004700822290033102, "rewards/judge_quality/mean": 0.30124998092651367, "rewards/judge_quality/std": 0.10398317128419876, "rewards/total_composite/mean": 0.6609954833984375, "rewards/total_composite/std": 0.19375428557395935, "reward": 0.6609954833984375, "reward_std": 0.19375428557395935, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1688535064458847, "sampling/sampling_logp_difference/max": 1.0816240310668945, "sampling/importance_sampling_ratio/min": 0.339044451713562, "sampling/importance_sampling_ratio/mean": 1.0338186025619507, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 2.1685265600681305, "clip_ratio/low_mean": 0.04505446366965771, "clip_ratio/low_min": 0.04505446366965771, "clip_ratio/high_mean": 0.12919956259429455, "clip_ratio/high_max": 0.12919956259429455, "clip_ratio/region_mean": 0.17425402626395226, "reward_total_mean": 0.6609954833984375, "reward_meter_mean": 0.7137892246246338, "reward_meter_std": 0.38705575466156006, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_hard_gate_mean": 1.0, "reward_hard_gate_std": 0.0, "reward_repeat_soft_mean": 0.9941529035568237, "reward_repeat_soft_std": 0.004700822290033102, "reward_judge_quality_mean": 0.30124998092651367, "reward_judge_quality_std": 0.10398317128419876, "reward_total_composite_mean": 0.6609954833984375, "reward_total_composite_std": 0.19375428557395935}