{"timestamp_utc": "2026-04-11T22:36:55Z", "mode": "train", "global_step": 651, "epoch": 0.026147728642005062, "loss": -0.0071, "grad_norm": 9.100071907043457, "learning_rate": 8.03030303030303e-06, "num_tokens": 1409555.0, "completions/mean_length": 35.125, "completions/min_length": 33.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.125, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9935092329978943, "rewards/meter/std": 0.000985646271146834, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9935092329978943, "rewards/total_composite/std": 0.000985646271146834, "reward": 0.9935092329978943, "reward_std": 0.0009856420801952481, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04576735198497772, "sampling/sampling_logp_difference/max": 0.9458228349685669, "sampling/importance_sampling_ratio/min": 0.3883599042892456, "sampling/importance_sampling_ratio/mean": 1.011391282081604, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.17123969458043575, "clip_ratio/low_mean": 0.028315248200669885, "clip_ratio/low_min": 0.028315248200669885, "clip_ratio/high_mean": 0.010521235642954707, "clip_ratio/high_max": 0.010521235642954707, "clip_ratio/region_mean": 0.03883648384362459, "reward_total_mean": 0.9935092329978943, "reward_meter_mean": 0.9935092329978943, "reward_meter_std": 0.000985646271146834, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9935092329978943, "reward_total_composite_std": 0.000985646271146834} {"timestamp_utc": "2026-04-11T22:37:05Z", "mode": "train", "global_step": 652, "epoch": 0.026187894123790016, "loss": -0.2894, "grad_norm": 0.4141400456428528, "learning_rate": 8.027272727272728e-06, "num_tokens": 1413749.0, "completions/mean_length": 374.25, "completions/min_length": 343.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 354.5714416503906, "completions/min_terminated_length": 343.0, "completions/max_terminated_length": 363.0, "rewards/meter/mean": 0.8732885122299194, "rewards/meter/std": 0.3528631627559662, "rewards/count_adherence/mean": 0.824999988079071, "rewards/count_adherence/std": 0.337003618478775, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.17251461744308472, "rewards/repeat_penalty/std": 0.33435773849487305, "rewards/total_composite/mean": 0.04464861750602722, "rewards/total_composite/std": 0.018085261806845665, "reward": 0.04464861750602722, "reward_std": 0.018085261806845665, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004473666660487652, "sampling/sampling_logp_difference/max": 0.7091238498687744, "sampling/importance_sampling_ratio/min": 0.4920751452445984, "sampling/importance_sampling_ratio/mean": 1.0013567209243774, "sampling/importance_sampling_ratio/max": 1.834702491760254, "entropy": 0.019337893230840564, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0035446555411908776, "clip_ratio/high_max": 0.0035446555411908776, "clip_ratio/region_mean": 0.0035446555411908776, "reward_total_mean": 0.04464861750602722, "reward_meter_mean": 0.8732885122299194, "reward_meter_std": 0.3528631627559662, "reward_count_adherence_mean": 0.824999988079071, "reward_count_adherence_std": 0.337003618478775, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.17251461744308472, "reward_repeat_penalty_std": 0.33435773849487305, "reward_total_composite_mean": 0.04464861750602722, "reward_total_composite_std": 0.018085261806845665} {"timestamp_utc": "2026-04-11T22:37:15Z", "mode": "train", "global_step": 653, "epoch": 0.02622805960557497, "loss": 0.1416, "grad_norm": 1.191830039024353, "learning_rate": 8.024242424242425e-06, "num_tokens": 1415617.0, "completions/mean_length": 133.5, "completions/min_length": 78.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 79.42857360839844, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.9687988758087158, "rewards/meter/std": 0.06074121594429016, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.4583333432674408, "rewards/repeat_penalty/std": 0.24800792336463928, "rewards/total_composite/mean": 0.43250638246536255, "rewards/total_composite/std": 0.1944577693939209, "reward": 0.43250638246536255, "reward_std": 0.1944577544927597, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014217361807823181, "sampling/sampling_logp_difference/max": 0.9473090171813965, "sampling/importance_sampling_ratio/min": 0.3877831697463989, "sampling/importance_sampling_ratio/mean": 1.0012526512145996, "sampling/importance_sampling_ratio/max": 1.5146371126174927, "entropy": 0.0790142323821783, "clip_ratio/low_mean": 0.007873826543800533, "clip_ratio/low_min": 0.007873826543800533, "clip_ratio/high_mean": 0.0029761905316263437, "clip_ratio/high_max": 0.0029761905316263437, "clip_ratio/region_mean": 0.010850017075426877, "reward_total_mean": 0.43250638246536255, "reward_meter_mean": 0.9687988758087158, "reward_meter_std": 0.06074121594429016, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.4583333432674408, "reward_repeat_penalty_std": 0.24800792336463928, "reward_total_composite_mean": 0.43250638246536255, "reward_total_composite_std": 0.1944577693939209} {"timestamp_utc": "2026-04-11T22:37:21Z", "mode": "train", "global_step": 654, "epoch": 0.026268225087359924, "loss": -0.0232, "grad_norm": 1.927111268043518, "learning_rate": 8.021212121212122e-06, "num_tokens": 1418478.0, "completions/mean_length": 170.625, "completions/min_length": 162.0, "completions/max_length": 183.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 170.625, "completions/min_terminated_length": 162.0, "completions/max_terminated_length": 183.0, "rewards/meter/mean": 0.9953961968421936, "rewards/meter/std": 0.0023201238363981247, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.2678571343421936, "rewards/repeat_penalty/std": 0.17806050181388855, "rewards/total_composite/mean": 0.2665513753890991, "rewards/total_composite/std": 0.17725923657417297, "reward": 0.2665513753890991, "reward_std": 0.17725922167301178, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01804671622812748, "sampling/sampling_logp_difference/max": 1.4201292991638184, "sampling/importance_sampling_ratio/min": 0.24168278276920319, "sampling/importance_sampling_ratio/mean": 1.0009305477142334, "sampling/importance_sampling_ratio/max": 1.6224781274795532, "entropy": 0.08471588138490915, "clip_ratio/low_mean": 0.003801907878369093, "clip_ratio/low_min": 0.003801907878369093, "clip_ratio/high_mean": 0.008620842476375401, "clip_ratio/high_max": 0.008620842476375401, "clip_ratio/region_mean": 0.012422750354744494, "reward_total_mean": 0.2665513753890991, "reward_meter_mean": 0.9953961968421936, "reward_meter_std": 0.0023201238363981247, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.2678571343421936, "reward_repeat_penalty_std": 0.17806050181388855, "reward_total_composite_mean": 0.2665513753890991, "reward_total_composite_std": 0.17725923657417297} {"timestamp_utc": "2026-04-11T22:37:30Z", "mode": "train", "global_step": 655, "epoch": 0.026308390569144878, "loss": -0.0945, "grad_norm": 2.2248940467834473, "learning_rate": 8.018181818181818e-06, "num_tokens": 1420163.0, "completions/mean_length": 116.625, "completions/min_length": 58.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 60.142860412597656, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.7503823637962341, "rewards/meter/std": 0.44998785853385925, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.5, "rewards/repeat_penalty/std": 0.30860671401023865, "rewards/total_composite/mean": 0.25364238023757935, "rewards/total_composite/std": 0.14388985931873322, "reward": 0.25364238023757935, "reward_std": 0.1438898742198944, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024109482765197754, "sampling/sampling_logp_difference/max": 1.0410680770874023, "sampling/importance_sampling_ratio/min": 0.35307735204696655, "sampling/importance_sampling_ratio/mean": 1.0044450759887695, "sampling/importance_sampling_ratio/max": 1.6566717624664307, "entropy": 0.17866009753197432, "clip_ratio/low_mean": 0.0018382353009656072, "clip_ratio/low_min": 0.0018382353009656072, "clip_ratio/high_mean": 0.012714299838989973, "clip_ratio/high_max": 0.012714299838989973, "clip_ratio/region_mean": 0.01455253513995558, "reward_total_mean": 0.25364238023757935, "reward_meter_mean": 0.7503823637962341, "reward_meter_std": 0.44998785853385925, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.5, "reward_repeat_penalty_std": 0.30860671401023865, "reward_total_composite_mean": 0.25364238023757935, "reward_total_composite_std": 0.14388985931873322} {"timestamp_utc": "2026-04-11T22:37:35Z", "mode": "train", "global_step": 656, "epoch": 0.026348556050929832, "loss": 0.0094, "grad_norm": 6.352079391479492, "learning_rate": 8.015151515151515e-06, "num_tokens": 1421891.0, "completions/mean_length": 58.0, "completions/min_length": 55.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9239930510520935, "rewards/meter/std": 0.15441425144672394, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.3333333432674408, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.3079977035522461, "rewards/total_composite/std": 0.051471415907144547, "reward": 0.3079977035522461, "reward_std": 0.05147142335772514, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022419268265366554, "sampling/sampling_logp_difference/max": 0.8770105838775635, "sampling/importance_sampling_ratio/min": 0.41602474451065063, "sampling/importance_sampling_ratio/mean": 1.0036613941192627, "sampling/importance_sampling_ratio/max": 1.928532361984253, "entropy": 0.16807558294385672, "clip_ratio/low_mean": 0.004310344811528921, "clip_ratio/low_min": 0.004310344811528921, "clip_ratio/high_mean": 0.01278492109850049, "clip_ratio/high_max": 0.01278492109850049, "clip_ratio/region_mean": 0.01709526591002941, "reward_total_mean": 0.3079977035522461, "reward_meter_mean": 0.9239930510520935, "reward_meter_std": 0.15441425144672394, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.3333333432674408, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.3079977035522461, "reward_total_composite_std": 0.051471415907144547} {"timestamp_utc": "2026-04-11T22:37:41Z", "mode": "train", "global_step": 657, "epoch": 0.026388721532714786, "loss": 0.0154, "grad_norm": 2.9878616333007812, "learning_rate": 8.012121212121214e-06, "num_tokens": 1424909.0, "completions/mean_length": 181.25, "completions/min_length": 175.0, "completions/max_length": 185.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 181.25, "completions/min_terminated_length": 175.0, "completions/max_terminated_length": 185.0, "rewards/meter/mean": 0.9874005317687988, "rewards/meter/std": 0.011436098255217075, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.20192307233810425, "rewards/repeat_penalty/std": 0.2094048410654068, "rewards/total_composite/mean": 0.1662987768650055, "rewards/total_composite/std": 0.17312301695346832, "reward": 0.1662987768650055, "reward_std": 0.17312301695346832, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010927603580057621, "sampling/sampling_logp_difference/max": 1.7186641693115234, "sampling/importance_sampling_ratio/min": 0.1793055236339569, "sampling/importance_sampling_ratio/mean": 1.0016026496887207, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.042974324664101005, "clip_ratio/low_mean": 0.008925562433432788, "clip_ratio/low_min": 0.008925562433432788, "clip_ratio/high_mean": 0.0028169237775728106, "clip_ratio/high_max": 0.0028169237775728106, "clip_ratio/region_mean": 0.011742486211005598, "reward_total_mean": 0.1662987768650055, "reward_meter_mean": 0.9874005317687988, "reward_meter_std": 0.011436098255217075, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.20192307233810425, "reward_repeat_penalty_std": 0.2094048410654068, "reward_total_composite_mean": 0.1662987768650055, "reward_total_composite_std": 0.17312301695346832} {"timestamp_utc": "2026-04-11T22:37:51Z", "mode": "train", "global_step": 658, "epoch": 0.02642888701449974, "loss": -0.0762, "grad_norm": 0.7183297276496887, "learning_rate": 8.00909090909091e-06, "num_tokens": 1429785.0, "completions/mean_length": 428.5, "completions/min_length": 394.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 416.5714416503906, "completions/min_terminated_length": 394.0, "completions/max_terminated_length": 441.0, "rewards/meter/mean": 0.8735865354537964, "rewards/meter/std": 0.35298284888267517, "rewards/count_adherence/mean": 0.6416666507720947, "rewards/count_adherence/std": 0.26170989871025085, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.17413419485092163, "rewards/repeat_penalty/std": 0.34882497787475586, "rewards/total_composite/mean": 0.033445145934820175, "rewards/total_composite/std": 0.06880706548690796, "reward": 0.033445145934820175, "reward_std": 0.06880706548690796, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0026275559794157743, "sampling/sampling_logp_difference/max": 1.0090997219085693, "sampling/importance_sampling_ratio/min": 0.3645470440387726, "sampling/importance_sampling_ratio/mean": 1.0009920597076416, "sampling/importance_sampling_ratio/max": 1.7735869884490967, "entropy": 0.012874894309788942, "clip_ratio/low_mean": 0.00030637255986221135, "clip_ratio/low_min": 0.00030637255986221135, "clip_ratio/high_mean": 0.0005817760829813778, "clip_ratio/high_max": 0.0005817760829813778, "clip_ratio/region_mean": 0.0008881486428435892, "reward_total_mean": 0.033445145934820175, "reward_meter_mean": 0.8735865354537964, "reward_meter_std": 0.35298284888267517, "reward_count_adherence_mean": 0.6416666507720947, "reward_count_adherence_std": 0.26170989871025085, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.17413419485092163, "reward_repeat_penalty_std": 0.34882497787475586, "reward_total_composite_mean": 0.033445145934820175, "reward_total_composite_std": 0.06880706548690796} {"timestamp_utc": "2026-04-11T22:38:00Z", "mode": "train", "global_step": 659, "epoch": 0.026469052496284694, "loss": -0.1652, "grad_norm": 1.3609012365341187, "learning_rate": 8.006060606060607e-06, "num_tokens": 1431540.0, "completions/mean_length": 190.375, "completions/min_length": 81.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 83.16667175292969, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 86.0, "rewards/meter/mean": 0.9505882263183594, "rewards/meter/std": 0.030698692426085472, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/repeat_penalty/mean": 0.7916666865348816, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.513268232345581, "rewards/total_composite/std": 0.3378344178199768, "reward": 0.513268232345581, "reward_std": 0.3378343880176544, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02022649347782135, "sampling/sampling_logp_difference/max": 1.8950445652008057, "sampling/importance_sampling_ratio/min": 0.1503116339445114, "sampling/importance_sampling_ratio/mean": 1.0079931020736694, "sampling/importance_sampling_ratio/max": 1.7507625818252563, "entropy": 0.06744233565405011, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.008969587041065097, "clip_ratio/high_max": 0.008969587041065097, "clip_ratio/region_mean": 0.008969587041065097, "reward_total_mean": 0.513268232345581, "reward_meter_mean": 0.9505882263183594, "reward_meter_std": 0.030698692426085472, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_repeat_penalty_mean": 0.7916666865348816, "reward_repeat_penalty_std": 0.17251639068126678, "reward_total_composite_mean": 0.513268232345581, "reward_total_composite_std": 0.3378344178199768} {"timestamp_utc": "2026-04-11T22:38:05Z", "mode": "train", "global_step": 660, "epoch": 0.026509217978069648, "loss": 0.0308, "grad_norm": 2.2579872608184814, "learning_rate": 8.003030303030304e-06, "num_tokens": 1433681.0, "completions/mean_length": 110.625, "completions/min_length": 104.0, "completions/max_length": 138.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 110.625, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 138.0, "rewards/meter/mean": 0.9927444458007812, "rewards/meter/std": 0.0012347318697720766, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.23571428656578064, "rewards/repeat_penalty/std": 0.07284314185380936, "rewards/total_composite/mean": 0.22219520807266235, "rewards/total_composite/std": 0.07082200050354004, "reward": 0.22219520807266235, "reward_std": 0.07082199305295944, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008687403053045273, "sampling/sampling_logp_difference/max": 1.8446245193481445, "sampling/importance_sampling_ratio/min": 0.1580846756696701, "sampling/importance_sampling_ratio/mean": 1.001028299331665, "sampling/importance_sampling_ratio/max": 1.6747108697891235, "entropy": 0.020129066659137607, "clip_ratio/low_mean": 0.006200430449098349, "clip_ratio/low_min": 0.006200430449098349, "clip_ratio/high_mean": 0.0036057692486792803, "clip_ratio/high_max": 0.0036057692486792803, "clip_ratio/region_mean": 0.009806199697777629, "reward_total_mean": 0.22219520807266235, "reward_meter_mean": 0.9927444458007812, "reward_meter_std": 0.0012347318697720766, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.23571428656578064, "reward_repeat_penalty_std": 0.07284314185380936, "reward_total_composite_mean": 0.22219520807266235, "reward_total_composite_std": 0.07082200050354004} {"timestamp_utc": "2026-04-11T22:38:15Z", "mode": "train", "global_step": 661, "epoch": 0.0265493834598546, "loss": -0.0515, "grad_norm": 3.181408405303955, "learning_rate": 8.000000000000001e-06, "num_tokens": 1435750.0, "completions/mean_length": 203.625, "completions/min_length": 81.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 100.83333587646484, "completions/min_terminated_length": 81.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.43306416273117065, "rewards/meter/std": 0.44791871309280396, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.18600596487522125, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/repeat_penalty/mean": 0.6357142925262451, "rewards/repeat_penalty/std": 0.31916436553001404, "rewards/total_composite/mean": 0.22007080912590027, "rewards/total_composite/std": 0.2968771755695343, "reward": 0.22007080912590027, "reward_std": 0.2968771457672119, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04375706985592842, "sampling/sampling_logp_difference/max": 1.372495174407959, "sampling/importance_sampling_ratio/min": 0.253473699092865, "sampling/importance_sampling_ratio/mean": 1.0068386793136597, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.17050567921251059, "clip_ratio/low_mean": 0.008919409476220608, "clip_ratio/low_min": 0.008919409476220608, "clip_ratio/high_mean": 0.017718179151415825, "clip_ratio/high_max": 0.017718179151415825, "clip_ratio/region_mean": 0.026637588627636433, "reward_total_mean": 0.22007080912590027, "reward_meter_mean": 0.43306416273117065, "reward_meter_std": 0.44791871309280396, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.18600596487522125, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_repeat_penalty_mean": 0.6357142925262451, "reward_repeat_penalty_std": 0.31916436553001404, "reward_total_composite_mean": 0.22007080912590027, "reward_total_composite_std": 0.2968771755695343} {"timestamp_utc": "2026-04-11T22:38:25Z", "mode": "train", "global_step": 662, "epoch": 0.026589548941639556, "loss": -0.1279, "grad_norm": 2.6338069438934326, "learning_rate": 7.996969696969697e-06, "num_tokens": 1437638.0, "completions/mean_length": 117.0, "completions/min_length": 57.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 60.57143020629883, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9266777634620667, "rewards/meter/std": 0.1883465051651001, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.5833333730697632, "rewards/repeat_penalty/std": 0.2357022762298584, "rewards/total_composite/mean": 0.455108642578125, "rewards/total_composite/std": 0.24599777162075043, "reward": 0.455108642578125, "reward_std": 0.24599777162075043, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.032812319695949554, "sampling/sampling_logp_difference/max": 1.3324346542358398, "sampling/importance_sampling_ratio/min": 0.263834148645401, "sampling/importance_sampling_ratio/mean": 1.0021228790283203, "sampling/importance_sampling_ratio/max": 1.474334478378296, "entropy": 0.17267457023262978, "clip_ratio/low_mean": 0.004385964944958687, "clip_ratio/low_min": 0.004385964944958687, "clip_ratio/high_mean": 0.012073024990968406, "clip_ratio/high_max": 0.012073024990968406, "clip_ratio/region_mean": 0.016458989935927093, "reward_total_mean": 0.455108642578125, "reward_meter_mean": 0.9266777634620667, "reward_meter_std": 0.1883465051651001, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.5833333730697632, "reward_repeat_penalty_std": 0.2357022762298584, "reward_total_composite_mean": 0.455108642578125, "reward_total_composite_std": 0.24599777162075043} {"timestamp_utc": "2026-04-11T22:38:30Z", "mode": "train", "global_step": 663, "epoch": 0.02662971442342451, "loss": 0.021, "grad_norm": 4.197593688964844, "learning_rate": 7.993939393939396e-06, "num_tokens": 1439950.0, "completions/mean_length": 108.0, "completions/min_length": 96.0, "completions/max_length": 133.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.0, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 133.0, "rewards/meter/mean": 0.9684537053108215, "rewards/meter/std": 0.027583837509155273, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.4226190447807312, "rewards/repeat_penalty/std": 0.2210753709077835, "rewards/total_composite/mean": 0.3900730013847351, "rewards/total_composite/std": 0.19891957938671112, "reward": 0.3900730013847351, "reward_std": 0.19891956448554993, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02085985243320465, "sampling/sampling_logp_difference/max": 1.0295929908752441, "sampling/importance_sampling_ratio/min": 0.3571523129940033, "sampling/importance_sampling_ratio/mean": 0.9997091889381409, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11175651382654905, "clip_ratio/low_mean": 0.00461137923412025, "clip_ratio/low_min": 0.00461137923412025, "clip_ratio/high_mean": 0.014703674940392375, "clip_ratio/high_max": 0.014703674940392375, "clip_ratio/region_mean": 0.019315054174512625, "reward_total_mean": 0.3900730013847351, "reward_meter_mean": 0.9684537053108215, "reward_meter_std": 0.027583837509155273, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.4226190447807312, "reward_repeat_penalty_std": 0.2210753709077835, "reward_total_composite_mean": 0.3900730013847351, "reward_total_composite_std": 0.19891957938671112} {"timestamp_utc": "2026-04-11T22:38:35Z", "mode": "train", "global_step": 664, "epoch": 0.026669879905209463, "loss": -0.029, "grad_norm": 7.99793004989624, "learning_rate": 7.990909090909091e-06, "num_tokens": 1441763.0, "completions/mean_length": 70.625, "completions/min_length": 68.0, "completions/max_length": 79.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.625, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.9828510880470276, "rewards/meter/std": 0.025825461372733116, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5833333730697632, "rewards/repeat_penalty/std": 0.29546841979026794, "rewards/total_composite/mean": 0.568142294883728, "rewards/total_composite/std": 0.27455055713653564, "reward": 0.568142294883728, "reward_std": 0.27455058693885803, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03234035521745682, "sampling/sampling_logp_difference/max": 1.1230463981628418, "sampling/importance_sampling_ratio/min": 0.3252873122692108, "sampling/importance_sampling_ratio/mean": 1.0028775930404663, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10323597816750407, "clip_ratio/low_mean": 0.0055147059028968215, "clip_ratio/low_min": 0.0055147059028968215, "clip_ratio/high_mean": 0.023867564275860786, "clip_ratio/high_max": 0.023867564275860786, "clip_ratio/region_mean": 0.029382270178757608, "reward_total_mean": 0.568142294883728, "reward_meter_mean": 0.9828510880470276, "reward_meter_std": 0.025825461372733116, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5833333730697632, "reward_repeat_penalty_std": 0.29546841979026794, "reward_total_composite_mean": 0.568142294883728, "reward_total_composite_std": 0.27455055713653564} {"timestamp_utc": "2026-04-11T22:38:39Z", "mode": "train", "global_step": 665, "epoch": 0.026710045386994417, "loss": -0.0097, "grad_norm": 3.2300848960876465, "learning_rate": 7.987878787878789e-06, "num_tokens": 1443616.0, "completions/mean_length": 80.625, "completions/min_length": 74.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.625, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.9603564143180847, "rewards/meter/std": 0.023673059418797493, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7916666865348816, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.7601395845413208, "rewards/total_composite/std": 0.16597026586532593, "reward": 0.7601395845413208, "reward_std": 0.16597026586532593, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0124077582731843, "sampling/sampling_logp_difference/max": 1.0308361053466797, "sampling/importance_sampling_ratio/min": 0.3567086160182953, "sampling/importance_sampling_ratio/mean": 1.0014386177062988, "sampling/importance_sampling_ratio/max": 1.564980387687683, "entropy": 0.07347342604771256, "clip_ratio/low_mean": 0.009405238670296967, "clip_ratio/low_min": 0.009405238670296967, "clip_ratio/high_mean": 0.0015432098880410194, "clip_ratio/high_max": 0.0015432098880410194, "clip_ratio/region_mean": 0.010948448558337986, "reward_total_mean": 0.7601395845413208, "reward_meter_mean": 0.9603564143180847, "reward_meter_std": 0.023673059418797493, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7916666865348816, "reward_repeat_penalty_std": 0.17251639068126678, "reward_total_composite_mean": 0.7601395845413208, "reward_total_composite_std": 0.16597026586532593} {"timestamp_utc": "2026-04-11T22:38:44Z", "mode": "train", "global_step": 666, "epoch": 0.02675021086877937, "loss": 0.0007, "grad_norm": 3.160210609436035, "learning_rate": 7.984848484848486e-06, "num_tokens": 1445930.0, "completions/mean_length": 122.25, "completions/min_length": 120.0, "completions/max_length": 135.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.25, "completions/min_terminated_length": 120.0, "completions/max_terminated_length": 135.0, "rewards/meter/mean": 0.8794082999229431, "rewards/meter/std": 0.32782238721847534, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.212053582072258, "rewards/repeat_penalty/std": 0.2030286192893982, "rewards/total_composite/mean": 0.12827160954475403, "rewards/total_composite/std": 0.03277355059981346, "reward": 0.12827160954475403, "reward_std": 0.03277355059981346, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02072002924978733, "sampling/sampling_logp_difference/max": 6.419788360595703, "sampling/importance_sampling_ratio/min": 0.0016290009953081608, "sampling/importance_sampling_ratio/mean": 0.9980396032333374, "sampling/importance_sampling_ratio/max": 1.6665557622909546, "entropy": 0.03877314692363143, "clip_ratio/low_mean": 0.0010416667209938169, "clip_ratio/low_min": 0.0010416667209938169, "clip_ratio/high_mean": 0.006232782383449376, "clip_ratio/high_max": 0.006232782383449376, "clip_ratio/region_mean": 0.0072744491044431925, "reward_total_mean": 0.12827160954475403, "reward_meter_mean": 0.8794082999229431, "reward_meter_std": 0.32782238721847534, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.212053582072258, "reward_repeat_penalty_std": 0.2030286192893982, "reward_total_composite_mean": 0.12827160954475403, "reward_total_composite_std": 0.03277355059981346} {"timestamp_utc": "2026-04-11T22:38:50Z", "mode": "train", "global_step": 667, "epoch": 0.026790376350564325, "loss": 0.0108, "grad_norm": 4.775012016296387, "learning_rate": 7.981818181818183e-06, "num_tokens": 1448313.0, "completions/mean_length": 120.875, "completions/min_length": 116.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 120.875, "completions/min_terminated_length": 116.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.9965612888336182, "rewards/meter/std": 0.0018690497381612659, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.25, "rewards/repeat_penalty/std": 0.21257823705673218, "rewards/total_composite/mean": 0.24911293387413025, "rewards/total_composite/std": 0.2116239219903946, "reward": 0.24911293387413025, "reward_std": 0.2116239219903946, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017011119052767754, "sampling/sampling_logp_difference/max": 1.0500693321228027, "sampling/importance_sampling_ratio/min": 0.3773258924484253, "sampling/importance_sampling_ratio/mean": 0.9989076256752014, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07529849279671907, "clip_ratio/low_mean": 0.0031968391267582774, "clip_ratio/low_min": 0.0031968391267582774, "clip_ratio/high_mean": 0.006372183095663786, "clip_ratio/high_max": 0.006372183095663786, "clip_ratio/region_mean": 0.009569022222422063, "reward_total_mean": 0.24911293387413025, "reward_meter_mean": 0.9965612888336182, "reward_meter_std": 0.0018690497381612659, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.25, "reward_repeat_penalty_std": 0.21257823705673218, "reward_total_composite_mean": 0.24911293387413025, "reward_total_composite_std": 0.2116239219903946} {"timestamp_utc": "2026-04-11T22:38:59Z", "mode": "train", "global_step": 668, "epoch": 0.02683054183234928, "loss": -0.1559, "grad_norm": 2.0396084785461426, "learning_rate": 7.978787878787879e-06, "num_tokens": 1451329.0, "completions/mean_length": 243.0, "completions/min_length": 195.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 204.57144165039062, "completions/min_terminated_length": 195.0, "completions/max_terminated_length": 207.0, "rewards/meter/mean": 0.9733182787895203, "rewards/meter/std": 0.02887692302465439, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.27272728085517883, "rewards/repeat_penalty/std": 0.13744163513183594, "rewards/total_composite/mean": 0.17570531368255615, "rewards/total_composite/std": 0.11893084645271301, "reward": 0.17570531368255615, "reward_std": 0.11893083900213242, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01857808604836464, "sampling/sampling_logp_difference/max": 2.0781538486480713, "sampling/importance_sampling_ratio/min": 0.1251610666513443, "sampling/importance_sampling_ratio/mean": 0.9996070861816406, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.05400602100417018, "clip_ratio/low_mean": 0.007991038146428764, "clip_ratio/low_min": 0.007991038146428764, "clip_ratio/high_mean": 0.0030251864809542894, "clip_ratio/high_max": 0.0030251864809542894, "clip_ratio/region_mean": 0.011016224627383053, "reward_total_mean": 0.17570531368255615, "reward_meter_mean": 0.9733182787895203, "reward_meter_std": 0.02887692302465439, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.27272728085517883, "reward_repeat_penalty_std": 0.13744163513183594, "reward_total_composite_mean": 0.17570531368255615, "reward_total_composite_std": 0.11893084645271301} {"timestamp_utc": "2026-04-11T22:39:09Z", "mode": "train", "global_step": 669, "epoch": 0.026870707314134233, "loss": -0.023, "grad_norm": 2.1599395275115967, "learning_rate": 7.975757575757576e-06, "num_tokens": 1453130.0, "completions/mean_length": 240.125, "completions/min_length": 74.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 77.0, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.5011062026023865, "rewards/meter/std": 0.4051038920879364, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4261804223060608, "rewards/total_composite/std": 0.4667132496833801, "reward": 0.4261804223060608, "reward_std": 0.46671321988105774, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.052589599043130875, "sampling/sampling_logp_difference/max": 1.824247121810913, "sampling/importance_sampling_ratio/min": 0.16133907437324524, "sampling/importance_sampling_ratio/mean": 1.0080572366714478, "sampling/importance_sampling_ratio/max": 1.8760875463485718, "entropy": 0.16993126086890697, "clip_ratio/low_mean": 0.009377967799082398, "clip_ratio/low_min": 0.009377967799082398, "clip_ratio/high_mean": 0.011646514758467674, "clip_ratio/high_max": 0.011646514758467674, "clip_ratio/region_mean": 0.021024482557550073, "reward_total_mean": 0.4261804223060608, "reward_meter_mean": 0.5011062026023865, "reward_meter_std": 0.4051038920879364, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.4261804223060608, "reward_total_composite_std": 0.4667132496833801} {"timestamp_utc": "2026-04-11T22:39:18Z", "mode": "train", "global_step": 670, "epoch": 0.026910872795919187, "loss": -0.1072, "grad_norm": 0.766589879989624, "learning_rate": 7.972727272727273e-06, "num_tokens": 1454698.0, "completions/mean_length": 351.0, "completions/min_length": 73.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.625, "completions/mean_terminated_length": 82.66667175292969, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 93.0, "rewards/meter/mean": 0.6174333691596985, "rewards/meter/std": 0.4055008888244629, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.375, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/repeat_penalty/mean": 0.7083333730697632, "rewards/repeat_penalty/std": 0.11785111576318741, "rewards/total_composite/mean": 0.249087393283844, "rewards/total_composite/std": 0.3437734544277191, "reward": 0.249087393283844, "reward_std": 0.3437734842300415, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03491988405585289, "sampling/sampling_logp_difference/max": 1.1255141496658325, "sampling/importance_sampling_ratio/min": 0.4397118091583252, "sampling/importance_sampling_ratio/mean": 1.004920482635498, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09451978467404842, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.013186233583837748, "clip_ratio/high_max": 0.013186233583837748, "clip_ratio/region_mean": 0.013186233583837748, "reward_total_mean": 0.249087393283844, "reward_meter_mean": 0.6174333691596985, "reward_meter_std": 0.4055008888244629, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.375, "reward_arabic_clean_std": 0.5175492167472839, "reward_repeat_penalty_mean": 0.7083333730697632, "reward_repeat_penalty_std": 0.11785111576318741, "reward_total_composite_mean": 0.249087393283844, "reward_total_composite_std": 0.3437734544277191} {"timestamp_utc": "2026-04-11T22:39:28Z", "mode": "train", "global_step": 671, "epoch": 0.02695103827770414, "loss": -0.0912, "grad_norm": 0.7819209098815918, "learning_rate": 7.96969696969697e-06, "num_tokens": 1456328.0, "completions/mean_length": 354.75, "completions/min_length": 91.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.625, "completions/mean_terminated_length": 92.66667175292969, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 94.0, "rewards/meter/mean": 0.7872094511985779, "rewards/meter/std": 0.3365079164505005, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 0.5, "rewards/arabic_clean/std": 0.5345224738121033, "rewards/repeat_penalty/mean": 0.4750000238418579, "rewards/repeat_penalty/std": 0.2121320366859436, "rewards/total_composite/mean": 0.19879300892353058, "rewards/total_composite/std": 0.21251867711544037, "reward": 0.19879300892353058, "reward_std": 0.21251867711544037, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0315895713865757, "sampling/sampling_logp_difference/max": 1.163419246673584, "sampling/importance_sampling_ratio/min": 0.31241610646247864, "sampling/importance_sampling_ratio/mean": 1.0077892541885376, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07205967605113983, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.006735671544447541, "clip_ratio/high_max": 0.006735671544447541, "clip_ratio/region_mean": 0.006735671544447541, "reward_total_mean": 0.19879300892353058, "reward_meter_mean": 0.7872094511985779, "reward_meter_std": 0.3365079164505005, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 0.5, "reward_arabic_clean_std": 0.5345224738121033, "reward_repeat_penalty_mean": 0.4750000238418579, "reward_repeat_penalty_std": 0.2121320366859436, "reward_total_composite_mean": 0.19879300892353058, "reward_total_composite_std": 0.21251867711544037} {"timestamp_utc": "2026-04-11T22:39:38Z", "mode": "train", "global_step": 672, "epoch": 0.026991203759489095, "loss": -0.1237, "grad_norm": 0.8036666512489319, "learning_rate": 7.966666666666668e-06, "num_tokens": 1458728.0, "completions/mean_length": 314.0, "completions/min_length": 160.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 195.1999969482422, "completions/min_terminated_length": 160.0, "completions/max_terminated_length": 220.0, "rewards/meter/mean": 0.7117201685905457, "rewards/meter/std": 0.3770325481891632, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.1414213627576828, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/repeat_penalty/mean": 0.5681818127632141, "rewards/repeat_penalty/std": 0.2368127554655075, "rewards/total_composite/mean": 0.23602712154388428, "rewards/total_composite/std": 0.1749935895204544, "reward": 0.23602712154388428, "reward_std": 0.1749935895204544, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020951304584741592, "sampling/sampling_logp_difference/max": 1.4866762161254883, "sampling/importance_sampling_ratio/min": 0.22612299025058746, "sampling/importance_sampling_ratio/mean": 0.9994598627090454, "sampling/importance_sampling_ratio/max": 1.857919692993164, "entropy": 0.06148350611329079, "clip_ratio/low_mean": 0.0028696630615741014, "clip_ratio/low_min": 0.0028696630615741014, "clip_ratio/high_mean": 0.010866477387025952, "clip_ratio/high_max": 0.010866477387025952, "clip_ratio/region_mean": 0.013736140448600054, "reward_total_mean": 0.23602712154388428, "reward_meter_mean": 0.7117201685905457, "reward_meter_std": 0.3770325481891632, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.1414213627576828, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_repeat_penalty_mean": 0.5681818127632141, "reward_repeat_penalty_std": 0.2368127554655075, "reward_total_composite_mean": 0.23602712154388428, "reward_total_composite_std": 0.1749935895204544} {"timestamp_utc": "2026-04-11T22:39:48Z", "mode": "train", "global_step": 673, "epoch": 0.02703136924127405, "loss": -0.0271, "grad_norm": 4.668978214263916, "learning_rate": 7.963636363636365e-06, "num_tokens": 1460405.0, "completions/mean_length": 117.625, "completions/min_length": 58.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 61.28571701049805, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.8629885911941528, "rewards/meter/std": 0.28332778811454773, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.768899142742157, "rewards/total_composite/std": 0.3212806284427643, "reward": 0.768899142742157, "reward_std": 0.32128065824508667, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04510435834527016, "sampling/sampling_logp_difference/max": 1.6506894826889038, "sampling/importance_sampling_ratio/min": 0.1919175386428833, "sampling/importance_sampling_ratio/mean": 1.0077714920043945, "sampling/importance_sampling_ratio/max": 1.7451122999191284, "entropy": 0.2579981219023466, "clip_ratio/low_mean": 0.020833334419876337, "clip_ratio/low_min": 0.020833334419876337, "clip_ratio/high_mean": 0.02129602595232427, "clip_ratio/high_max": 0.02129602595232427, "clip_ratio/region_mean": 0.04212936037220061, "reward_total_mean": 0.768899142742157, "reward_meter_mean": 0.8629885911941528, "reward_meter_std": 0.28332778811454773, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.768899142742157, "reward_total_composite_std": 0.3212806284427643} {"timestamp_utc": "2026-04-11T22:39:58Z", "mode": "train", "global_step": 674, "epoch": 0.027071534723059003, "loss": -0.0956, "grad_norm": 3.691714286804199, "learning_rate": 7.96060606060606e-06, "num_tokens": 1462244.0, "completions/mean_length": 128.875, "completions/min_length": 69.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 74.14286041259766, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.7070336937904358, "rewards/meter/std": 0.39935943484306335, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.8333333730697632, "rewards/repeat_penalty/std": 0.17817415297031403, "rewards/total_composite/mean": 0.5691444873809814, "rewards/total_composite/std": 0.343730628490448, "reward": 0.5691444873809814, "reward_std": 0.3437305986881256, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05939958617091179, "sampling/sampling_logp_difference/max": 1.2720730304718018, "sampling/importance_sampling_ratio/min": 0.28025004267692566, "sampling/importance_sampling_ratio/mean": 1.0029557943344116, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.23062355443835258, "clip_ratio/low_mean": 0.009934040834195912, "clip_ratio/low_min": 0.009934040834195912, "clip_ratio/high_mean": 0.04135313397273421, "clip_ratio/high_max": 0.04135313397273421, "clip_ratio/region_mean": 0.051287174806930125, "reward_total_mean": 0.5691444873809814, "reward_meter_mean": 0.7070336937904358, "reward_meter_std": 0.39935943484306335, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.8333333730697632, "reward_repeat_penalty_std": 0.17817415297031403, "reward_total_composite_mean": 0.5691444873809814, "reward_total_composite_std": 0.343730628490448} {"timestamp_utc": "2026-04-11T22:40:08Z", "mode": "train", "global_step": 675, "epoch": 0.027111700204843957, "loss": -0.0973, "grad_norm": 0.7685384154319763, "learning_rate": 7.957575757575758e-06, "num_tokens": 1466421.0, "completions/mean_length": 376.125, "completions/min_length": 337.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 356.71429443359375, "completions/min_terminated_length": 337.0, "completions/max_terminated_length": 367.0, "rewards/meter/mean": 0.9768602848052979, "rewards/meter/std": 0.0202656090259552, "rewards/count_adherence/mean": 0.8624999523162842, "rewards/count_adherence/std": 0.31139087677001953, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.2991071343421936, "rewards/repeat_penalty/std": 0.36836329102516174, "rewards/total_composite/mean": 0.15836204588413239, "rewards/total_composite/std": 0.21933622658252716, "reward": 0.15836204588413239, "reward_std": 0.21933621168136597, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005116640590131283, "sampling/sampling_logp_difference/max": 1.467843770980835, "sampling/importance_sampling_ratio/min": 0.4014633893966675, "sampling/importance_sampling_ratio/mean": 1.0009857416152954, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0205289286095649, "clip_ratio/low_mean": 0.0032165506563615054, "clip_ratio/low_min": 0.0032165506563615054, "clip_ratio/high_mean": 0.000681198900565505, "clip_ratio/high_max": 0.000681198900565505, "clip_ratio/region_mean": 0.0038977495569270104, "reward_total_mean": 0.15836204588413239, "reward_meter_mean": 0.9768602848052979, "reward_meter_std": 0.0202656090259552, "reward_count_adherence_mean": 0.8624999523162842, "reward_count_adherence_std": 0.31139087677001953, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.2991071343421936, "reward_repeat_penalty_std": 0.36836329102516174, "reward_total_composite_mean": 0.15836204588413239, "reward_total_composite_std": 0.21933622658252716} {"timestamp_utc": "2026-04-11T22:40:12Z", "mode": "train", "global_step": 676, "epoch": 0.02715186568662891, "loss": -0.1218, "grad_norm": 9.489598274230957, "learning_rate": 7.954545454545455e-06, "num_tokens": 1468025.0, "completions/mean_length": 45.5, "completions/min_length": 41.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 45.5, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9916675090789795, "rewards/meter/std": 0.0027789692394435406, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9916675090789795, "rewards/total_composite/std": 0.0027789692394435406, "reward": 0.9916675090789795, "reward_std": 0.002778968308120966, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.040273357182741165, "sampling/sampling_logp_difference/max": 2.040062665939331, "sampling/importance_sampling_ratio/min": 0.13002057373523712, "sampling/importance_sampling_ratio/mean": 0.9964287877082825, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.16463111247867346, "clip_ratio/low_mean": 0.009073751280084252, "clip_ratio/low_min": 0.009073751280084252, "clip_ratio/high_mean": 0.01704174862243235, "clip_ratio/high_max": 0.01704174862243235, "clip_ratio/region_mean": 0.026115499902516603, "reward_total_mean": 0.9916675090789795, "reward_meter_mean": 0.9916675090789795, "reward_meter_std": 0.0027789692394435406, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9916675090789795, "reward_total_composite_std": 0.0027789692394435406} {"timestamp_utc": "2026-04-11T22:40:17Z", "mode": "train", "global_step": 677, "epoch": 0.027192031168413865, "loss": 0.0185, "grad_norm": 4.125744342803955, "learning_rate": 7.951515151515152e-06, "num_tokens": 1469690.0, "completions/mean_length": 62.125, "completions/min_length": 58.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.125, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9938265085220337, "rewards/meter/std": 0.0003937912406399846, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6666666865348816, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6625509858131409, "rewards/total_composite/std": 0.00026252749375998974, "reward": 0.6625509858131409, "reward_std": 0.0002625406195875257, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02882186695933342, "sampling/sampling_logp_difference/max": 5.798013687133789, "sampling/importance_sampling_ratio/min": 0.0030335744377225637, "sampling/importance_sampling_ratio/mean": 1.0004545450210571, "sampling/importance_sampling_ratio/max": 1.5561105012893677, "entropy": 0.11069364938884974, "clip_ratio/low_mean": 0.012098872568458319, "clip_ratio/low_min": 0.012098872568458319, "clip_ratio/high_mean": 0.0021551724057644606, "clip_ratio/high_max": 0.0021551724057644606, "clip_ratio/region_mean": 0.01425404497422278, "reward_total_mean": 0.6625509858131409, "reward_meter_mean": 0.9938265085220337, "reward_meter_std": 0.0003937912406399846, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6666666865348816, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6625509858131409, "reward_total_composite_std": 0.00026252749375998974} {"timestamp_utc": "2026-04-11T22:40:27Z", "mode": "train", "global_step": 678, "epoch": 0.02723219665019882, "loss": -0.15, "grad_norm": 0.7712501287460327, "learning_rate": 7.948484848484848e-06, "num_tokens": 1472788.0, "completions/mean_length": 336.25, "completions/min_length": 268.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 277.66668701171875, "completions/min_terminated_length": 268.0, "completions/max_terminated_length": 291.0, "rewards/meter/mean": 0.6280694007873535, "rewards/meter/std": 0.5047115683555603, "rewards/count_adherence/mean": 0.8035714626312256, "rewards/count_adherence/std": 0.15152288973331451, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/repeat_penalty/mean": 0.41633522510528564, "rewards/repeat_penalty/std": 0.3482770323753357, "rewards/total_composite/mean": 0.1397983729839325, "rewards/total_composite/std": 0.1796947717666626, "reward": 0.1397983729839325, "reward_std": 0.1796947568655014, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01377029623836279, "sampling/sampling_logp_difference/max": 5.731827259063721, "sampling/importance_sampling_ratio/min": 0.003241149475798011, "sampling/importance_sampling_ratio/mean": 0.9978663921356201, "sampling/importance_sampling_ratio/max": 1.6066981554031372, "entropy": 0.03139376197941601, "clip_ratio/low_mean": 0.004044867469929159, "clip_ratio/low_min": 0.004044867469929159, "clip_ratio/high_mean": 0.002251501166028902, "clip_ratio/high_max": 0.002251501166028902, "clip_ratio/region_mean": 0.006296368635958061, "reward_total_mean": 0.1397983729839325, "reward_meter_mean": 0.6280694007873535, "reward_meter_std": 0.5047115683555603, "reward_count_adherence_mean": 0.8035714626312256, "reward_count_adherence_std": 0.15152288973331451, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_repeat_penalty_mean": 0.41633522510528564, "reward_repeat_penalty_std": 0.3482770323753357, "reward_total_composite_mean": 0.1397983729839325, "reward_total_composite_std": 0.1796947717666626} {"timestamp_utc": "2026-04-11T22:40:31Z", "mode": "train", "global_step": 679, "epoch": 0.027272362131983773, "loss": -0.0123, "grad_norm": 12.517946243286133, "learning_rate": 7.945454545454547e-06, "num_tokens": 1474604.0, "completions/mean_length": 57.0, "completions/min_length": 53.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.0, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9841916561126709, "rewards/meter/std": 0.005828971043229103, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9428657293319702, "rewards/total_composite/std": 0.1139112040400505, "reward": 0.9428657293319702, "reward_std": 0.1139112189412117, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02025281824171543, "sampling/sampling_logp_difference/max": 1.1112747192382812, "sampling/importance_sampling_ratio/min": 0.3291391432285309, "sampling/importance_sampling_ratio/mean": 0.9998968243598938, "sampling/importance_sampling_ratio/max": 1.880238652229309, "entropy": 0.11628487333655357, "clip_ratio/low_mean": 0.002358490601181984, "clip_ratio/low_min": 0.002358490601181984, "clip_ratio/high_mean": 0.01512786210514605, "clip_ratio/high_max": 0.01512786210514605, "clip_ratio/region_mean": 0.017486352706328034, "reward_total_mean": 0.9428657293319702, "reward_meter_mean": 0.9841916561126709, "reward_meter_std": 0.005828971043229103, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.9428657293319702, "reward_total_composite_std": 0.1139112040400505} {"timestamp_utc": "2026-04-11T22:40:36Z", "mode": "train", "global_step": 680, "epoch": 0.027312527613768726, "loss": 0.031, "grad_norm": 8.33311653137207, "learning_rate": 7.942424242424242e-06, "num_tokens": 1476777.0, "completions/mean_length": 115.625, "completions/min_length": 101.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.625, "completions/min_terminated_length": 101.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.371232271194458, "rewards/meter/std": 0.47603076696395874, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.48125001788139343, "rewards/repeat_penalty/std": 0.13611315190792084, "rewards/total_composite/mean": 0.22119268774986267, "rewards/total_composite/std": 0.28687217831611633, "reward": 0.22119268774986267, "reward_std": 0.2868722081184387, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020960384979844093, "sampling/sampling_logp_difference/max": 1.9988516569137573, "sampling/importance_sampling_ratio/min": 0.1354907900094986, "sampling/importance_sampling_ratio/mean": 1.0001816749572754, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.08438334474340081, "clip_ratio/low_mean": 0.014545571291819215, "clip_ratio/low_min": 0.014545571291819215, "clip_ratio/high_mean": 0.0033385155256837606, "clip_ratio/high_max": 0.0033385155256837606, "clip_ratio/region_mean": 0.017884086817502975, "reward_total_mean": 0.22119268774986267, "reward_meter_mean": 0.371232271194458, "reward_meter_std": 0.47603076696395874, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.48125001788139343, "reward_repeat_penalty_std": 0.13611315190792084, "reward_total_composite_mean": 0.22119268774986267, "reward_total_composite_std": 0.28687217831611633} {"timestamp_utc": "2026-04-11T22:40:41Z", "mode": "train", "global_step": 681, "epoch": 0.02735269309555368, "loss": 0.0231, "grad_norm": 3.9222095012664795, "learning_rate": 7.93939393939394e-06, "num_tokens": 1478560.0, "completions/mean_length": 57.875, "completions/min_length": 57.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.875, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9723160266876221, "rewards/meter/std": 0.027899622917175293, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9166666865348816, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.8934845924377441, "rewards/total_composite/std": 0.1633896678686142, "reward": 0.8934845924377441, "reward_std": 0.163389652967453, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023344971239566803, "sampling/sampling_logp_difference/max": 1.0728743076324463, "sampling/importance_sampling_ratio/min": 0.342024028301239, "sampling/importance_sampling_ratio/mean": 1.0034892559051514, "sampling/importance_sampling_ratio/max": 1.6929512023925781, "entropy": 0.11835484858602285, "clip_ratio/low_mean": 0.014689265750348568, "clip_ratio/low_min": 0.014689265750348568, "clip_ratio/high_mean": 0.015127861872315407, "clip_ratio/high_max": 0.015127861872315407, "clip_ratio/region_mean": 0.029817127622663975, "reward_total_mean": 0.8934845924377441, "reward_meter_mean": 0.9723160266876221, "reward_meter_std": 0.027899622917175293, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9166666865348816, "reward_repeat_penalty_std": 0.15430334210395813, "reward_total_composite_mean": 0.8934845924377441, "reward_total_composite_std": 0.1633896678686142} {"timestamp_utc": "2026-04-11T22:40:50Z", "mode": "train", "global_step": 682, "epoch": 0.027392858577338634, "loss": -0.0676, "grad_norm": 1.2039419412612915, "learning_rate": 7.936363636363637e-06, "num_tokens": 1480663.0, "completions/mean_length": 173.875, "completions/min_length": 109.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 125.5714340209961, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 138.0, "rewards/meter/mean": 0.38672682642936707, "rewards/meter/std": 0.36102262139320374, "rewards/count_adherence/mean": 0.8333333730697632, "rewards/count_adherence/std": 0.2519763112068176, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.48750001192092896, "rewards/repeat_penalty/std": 0.24604006111621857, "rewards/total_composite/mean": 0.16035160422325134, "rewards/total_composite/std": 0.24009783565998077, "reward": 0.16035160422325134, "reward_std": 0.24009782075881958, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013726767152547836, "sampling/sampling_logp_difference/max": 1.2122516632080078, "sampling/importance_sampling_ratio/min": 0.29752659797668457, "sampling/importance_sampling_ratio/mean": 1.0000630617141724, "sampling/importance_sampling_ratio/max": 1.6329307556152344, "entropy": 0.049124513287097216, "clip_ratio/low_mean": 0.005251961061730981, "clip_ratio/low_min": 0.005251961061730981, "clip_ratio/high_mean": 0.0029069767333567142, "clip_ratio/high_max": 0.0029069767333567142, "clip_ratio/region_mean": 0.008158937795087695, "reward_total_mean": 0.16035160422325134, "reward_meter_mean": 0.38672682642936707, "reward_meter_std": 0.36102262139320374, "reward_count_adherence_mean": 0.8333333730697632, "reward_count_adherence_std": 0.2519763112068176, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.48750001192092896, "reward_repeat_penalty_std": 0.24604006111621857, "reward_total_composite_mean": 0.16035160422325134, "reward_total_composite_std": 0.24009783565998077} {"timestamp_utc": "2026-04-11T22:40:55Z", "mode": "train", "global_step": 683, "epoch": 0.02743302405912359, "loss": 0.0096, "grad_norm": 4.130276679992676, "learning_rate": 7.933333333333334e-06, "num_tokens": 1482349.0, "completions/mean_length": 58.75, "completions/min_length": 57.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.75, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9239417314529419, "rewards/meter/std": 0.05428864806890488, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9166666865348816, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.8474001884460449, "rewards/total_composite/std": 0.15459680557250977, "reward": 0.8474001884460449, "reward_std": 0.15459680557250977, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03419341892004013, "sampling/sampling_logp_difference/max": 1.8840758800506592, "sampling/importance_sampling_ratio/min": 0.15196943283081055, "sampling/importance_sampling_ratio/mean": 0.994130551815033, "sampling/importance_sampling_ratio/max": 1.6103789806365967, "entropy": 0.11382754053920507, "clip_ratio/low_mean": 0.0042372881434857845, "clip_ratio/low_min": 0.0042372881434857845, "clip_ratio/high_mean": 0.016894312808290124, "clip_ratio/high_max": 0.016894312808290124, "clip_ratio/region_mean": 0.02113160095177591, "reward_total_mean": 0.8474001884460449, "reward_meter_mean": 0.9239417314529419, "reward_meter_std": 0.05428864806890488, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9166666865348816, "reward_repeat_penalty_std": 0.15430334210395813, "reward_total_composite_mean": 0.8474001884460449, "reward_total_composite_std": 0.15459680557250977} {"timestamp_utc": "2026-04-11T22:40:59Z", "mode": "train", "global_step": 684, "epoch": 0.027473189540908542, "loss": -0.0064, "grad_norm": 4.445869445800781, "learning_rate": 7.930303030303031e-06, "num_tokens": 1484327.0, "completions/mean_length": 80.25, "completions/min_length": 79.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.25, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.9331251978874207, "rewards/meter/std": 0.07388782501220703, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5, "rewards/repeat_penalty/std": 0.10690450668334961, "rewards/total_composite/mean": 0.46393126249313354, "rewards/total_composite/std": 0.09499123692512512, "reward": 0.46393126249313354, "reward_std": 0.09499124437570572, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01861550658941269, "sampling/sampling_logp_difference/max": 1.1865754127502441, "sampling/importance_sampling_ratio/min": 0.3052648901939392, "sampling/importance_sampling_ratio/mean": 1.0021919012069702, "sampling/importance_sampling_ratio/max": 1.7777718305587769, "entropy": 0.0625843945890665, "clip_ratio/low_mean": 0.007758883642964065, "clip_ratio/low_min": 0.007758883642964065, "clip_ratio/high_mean": 0.004629629664123058, "clip_ratio/high_max": 0.004629629664123058, "clip_ratio/region_mean": 0.012388513307087123, "reward_total_mean": 0.46393126249313354, "reward_meter_mean": 0.9331251978874207, "reward_meter_std": 0.07388782501220703, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5, "reward_repeat_penalty_std": 0.10690450668334961, "reward_total_composite_mean": 0.46393126249313354, "reward_total_composite_std": 0.09499123692512512} {"timestamp_utc": "2026-04-11T22:41:05Z", "mode": "train", "global_step": 685, "epoch": 0.027513355022693496, "loss": 0.0202, "grad_norm": 2.300926685333252, "learning_rate": 7.927272727272729e-06, "num_tokens": 1487252.0, "completions/mean_length": 170.625, "completions/min_length": 165.0, "completions/max_length": 178.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 170.625, "completions/min_terminated_length": 165.0, "completions/max_terminated_length": 178.0, "rewards/meter/mean": 0.8037871718406677, "rewards/meter/std": 0.32751503586769104, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.4107142686843872, "rewards/repeat_penalty/std": 0.050507623702287674, "rewards/total_composite/mean": 0.34419798851013184, "rewards/total_composite/std": 0.14113976061344147, "reward": 0.34419798851013184, "reward_std": 0.14113974571228027, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010442078113555908, "sampling/sampling_logp_difference/max": 1.643384337425232, "sampling/importance_sampling_ratio/min": 0.19332465529441833, "sampling/importance_sampling_ratio/mean": 0.9983497262001038, "sampling/importance_sampling_ratio/max": 1.4589576721191406, "entropy": 0.046010758727788925, "clip_ratio/low_mean": 0.0007022471982054412, "clip_ratio/low_min": 0.0007022471982054412, "clip_ratio/high_mean": 0.010335821425542235, "clip_ratio/high_max": 0.010335821425542235, "clip_ratio/region_mean": 0.011038068623747677, "reward_total_mean": 0.34419798851013184, "reward_meter_mean": 0.8037871718406677, "reward_meter_std": 0.32751503586769104, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.4107142686843872, "reward_repeat_penalty_std": 0.050507623702287674, "reward_total_composite_mean": 0.34419798851013184, "reward_total_composite_std": 0.14113976061344147} {"timestamp_utc": "2026-04-11T22:41:15Z", "mode": "train", "global_step": 686, "epoch": 0.02755352050447845, "loss": -0.0998, "grad_norm": 1.047049641609192, "learning_rate": 7.924242424242426e-06, "num_tokens": 1490154.0, "completions/mean_length": 251.75, "completions/min_length": 212.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 214.57144165039062, "completions/min_terminated_length": 212.0, "completions/max_terminated_length": 222.0, "rewards/meter/mean": 0.9798320531845093, "rewards/meter/std": 0.047762468457221985, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.2828427255153656, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.32499998807907104, "rewards/repeat_penalty/std": 0.27645719051361084, "rewards/total_composite/mean": 0.2209167182445526, "rewards/total_composite/std": 0.04932519793510437, "reward": 0.2209167182445526, "reward_std": 0.04932519420981407, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011242986656725407, "sampling/sampling_logp_difference/max": 2.929636001586914, "sampling/importance_sampling_ratio/min": 0.053416479378938675, "sampling/importance_sampling_ratio/mean": 0.9986944198608398, "sampling/importance_sampling_ratio/max": 1.7827941179275513, "entropy": 0.021744283847510815, "clip_ratio/low_mean": 0.003458057180978358, "clip_ratio/low_min": 0.003458057180978358, "clip_ratio/high_mean": 0.00234195904340595, "clip_ratio/high_max": 0.00234195904340595, "clip_ratio/region_mean": 0.005800016224384308, "reward_total_mean": 0.2209167182445526, "reward_meter_mean": 0.9798320531845093, "reward_meter_std": 0.047762468457221985, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.2828427255153656, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.32499998807907104, "reward_repeat_penalty_std": 0.27645719051361084, "reward_total_composite_mean": 0.2209167182445526, "reward_total_composite_std": 0.04932519793510437} {"timestamp_utc": "2026-04-11T22:41:19Z", "mode": "train", "global_step": 687, "epoch": 0.027593685986263404, "loss": 0.0292, "grad_norm": 10.655113220214844, "learning_rate": 7.921212121212122e-06, "num_tokens": 1491874.0, "completions/mean_length": 71.0, "completions/min_length": 67.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.8845878839492798, "rewards/meter/std": 0.2909509837627411, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.8430095314979553, "rewards/total_composite/std": 0.2961682975292206, "reward": 0.8430095314979553, "reward_std": 0.2961682677268982, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05931292846798897, "sampling/sampling_logp_difference/max": 2.937185287475586, "sampling/importance_sampling_ratio/min": 0.05301474407315254, "sampling/importance_sampling_ratio/mean": 1.0036027431488037, "sampling/importance_sampling_ratio/max": 1.9307304620742798, "entropy": 0.1976525131613016, "clip_ratio/low_mean": 0.011955027701333165, "clip_ratio/low_min": 0.011955027701333165, "clip_ratio/high_mean": 0.020970338257029653, "clip_ratio/high_max": 0.020970338257029653, "clip_ratio/region_mean": 0.03292536595836282, "reward_total_mean": 0.8430095314979553, "reward_meter_mean": 0.8845878839492798, "reward_meter_std": 0.2909509837627411, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.8430095314979553, "reward_total_composite_std": 0.2961682975292206} {"timestamp_utc": "2026-04-11T22:41:24Z", "mode": "train", "global_step": 688, "epoch": 0.027633851468048358, "loss": -0.0086, "grad_norm": 1.4717203378677368, "learning_rate": 7.918181818181819e-06, "num_tokens": 1493736.0, "completions/mean_length": 67.75, "completions/min_length": 66.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.75, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.9984875321388245, "rewards/meter/std": 0.0001689638738753274, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6666666865348816, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6656583547592163, "rewards/total_composite/std": 0.00011263116175541654, "reward": 0.6656583547592163, "reward_std": 0.00011264239583397284, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018908310681581497, "sampling/sampling_logp_difference/max": 2.604221820831299, "sampling/importance_sampling_ratio/min": 0.07396066933870316, "sampling/importance_sampling_ratio/mean": 0.9985640048980713, "sampling/importance_sampling_ratio/max": 1.6862038373947144, "entropy": 0.06369170360267162, "clip_ratio/low_mean": 0.011363636702299118, "clip_ratio/low_min": 0.011363636702299118, "clip_ratio/high_mean": 0.010716472752392292, "clip_ratio/high_max": 0.010716472752392292, "clip_ratio/region_mean": 0.02208010945469141, "reward_total_mean": 0.6656583547592163, "reward_meter_mean": 0.9984875321388245, "reward_meter_std": 0.0001689638738753274, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6666666865348816, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6656583547592163, "reward_total_composite_std": 0.00011263116175541654} {"timestamp_utc": "2026-04-11T22:41:30Z", "mode": "train", "global_step": 689, "epoch": 0.027674016949833312, "loss": 0.0309, "grad_norm": 2.0693576335906982, "learning_rate": 7.915151515151516e-06, "num_tokens": 1497021.0, "completions/mean_length": 202.625, "completions/min_length": 191.0, "completions/max_length": 212.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 202.625, "completions/min_terminated_length": 191.0, "completions/max_terminated_length": 212.0, "rewards/meter/mean": 0.995306134223938, "rewards/meter/std": 0.0053658634424209595, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.4194444417953491, "rewards/repeat_penalty/std": 0.13975918292999268, "rewards/total_composite/mean": 0.4050329029560089, "rewards/total_composite/std": 0.1355670541524887, "reward": 0.4050329029560089, "reward_std": 0.1355670541524887, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014271133579313755, "sampling/sampling_logp_difference/max": 2.4142937660217285, "sampling/importance_sampling_ratio/min": 0.08943047374486923, "sampling/importance_sampling_ratio/mean": 0.9986175298690796, "sampling/importance_sampling_ratio/max": 1.674434781074524, "entropy": 0.03671248443424702, "clip_ratio/low_mean": 0.007414351915940642, "clip_ratio/low_min": 0.007414351915940642, "clip_ratio/high_mean": 0.0019430051324889064, "clip_ratio/high_max": 0.0019430051324889064, "clip_ratio/region_mean": 0.009357357048429549, "reward_total_mean": 0.4050329029560089, "reward_meter_mean": 0.995306134223938, "reward_meter_std": 0.0053658634424209595, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.4194444417953491, "reward_repeat_penalty_std": 0.13975918292999268, "reward_total_composite_mean": 0.4050329029560089, "reward_total_composite_std": 0.1355670541524887} {"timestamp_utc": "2026-04-11T22:41:39Z", "mode": "train", "global_step": 690, "epoch": 0.027714182431618266, "loss": 0.0024, "grad_norm": 1.1219055652618408, "learning_rate": 7.912121212121213e-06, "num_tokens": 1502074.0, "completions/mean_length": 401.625, "completions/min_length": 400.0, "completions/max_length": 402.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 401.625, "completions/min_terminated_length": 400.0, "completions/max_terminated_length": 402.0, "rewards/meter/mean": 0.9981971979141235, "rewards/meter/std": 0.000627980858553201, "rewards/count_adherence/mean": 0.7142857313156128, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.2503289580345154, "rewards/repeat_penalty/std": 0.23873279988765717, "rewards/total_composite/mean": 0.1784874051809311, "rewards/total_composite/std": 0.17026527225971222, "reward": 0.1784874051809311, "reward_std": 0.17026525735855103, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005141077097505331, "sampling/sampling_logp_difference/max": 1.3938775062561035, "sampling/importance_sampling_ratio/min": 0.24811138212680817, "sampling/importance_sampling_ratio/mean": 1.0005240440368652, "sampling/importance_sampling_ratio/max": 1.6704438924789429, "entropy": 0.027160495053976774, "clip_ratio/low_mean": 0.0021797263179905713, "clip_ratio/low_min": 0.0021797263179905713, "clip_ratio/high_mean": 0.0015578281017951667, "clip_ratio/high_max": 0.0015578281017951667, "clip_ratio/region_mean": 0.003737554419785738, "reward_total_mean": 0.1784874051809311, "reward_meter_mean": 0.9981971979141235, "reward_meter_std": 0.000627980858553201, "reward_count_adherence_mean": 0.7142857313156128, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.2503289580345154, "reward_repeat_penalty_std": 0.23873279988765717, "reward_total_composite_mean": 0.1784874051809311, "reward_total_composite_std": 0.17026527225971222} {"timestamp_utc": "2026-04-11T22:41:43Z", "mode": "train", "global_step": 691, "epoch": 0.02775434791340322, "loss": -0.1031, "grad_norm": 11.69098949432373, "learning_rate": 7.909090909090909e-06, "num_tokens": 1503707.0, "completions/mean_length": 38.125, "completions/min_length": 19.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.125, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.4370739161968231, "rewards/meter/std": 0.2700711488723755, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4370739161968231, "rewards/total_composite/std": 0.2700711488723755, "reward": 0.4370739161968231, "reward_std": 0.2700711488723755, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0835486352443695, "sampling/sampling_logp_difference/max": 1.102844476699829, "sampling/importance_sampling_ratio/min": 0.3319256007671356, "sampling/importance_sampling_ratio/mean": 1.0089696645736694, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5944820679724216, "clip_ratio/low_mean": 0.019354344811290503, "clip_ratio/low_min": 0.019354344811290503, "clip_ratio/high_mean": 0.04530784301459789, "clip_ratio/high_max": 0.04530784301459789, "clip_ratio/region_mean": 0.0646621878258884, "reward_total_mean": 0.4370739161968231, "reward_meter_mean": 0.4370739161968231, "reward_meter_std": 0.2700711488723755, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.4370739161968231, "reward_total_composite_std": 0.2700711488723755} {"timestamp_utc": "2026-04-11T22:41:47Z", "mode": "train", "global_step": 692, "epoch": 0.027794513395188174, "loss": 0.0153, "grad_norm": 5.031876564025879, "learning_rate": 7.906060606060608e-06, "num_tokens": 1505712.0, "completions/mean_length": 86.625, "completions/min_length": 84.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 86.625, "completions/min_terminated_length": 84.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.9942313432693481, "rewards/meter/std": 0.0012215422466397285, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9942313432693481, "rewards/total_composite/std": 0.0012215422466397285, "reward": 0.9942313432693481, "reward_std": 0.00122154806740582, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005889675114303827, "sampling/sampling_logp_difference/max": 0.44769424200057983, "sampling/importance_sampling_ratio/min": 0.7906866669654846, "sampling/importance_sampling_ratio/mean": 1.003043293952942, "sampling/importance_sampling_ratio/max": 1.5647001266479492, "entropy": 0.04205932654440403, "clip_ratio/low_mean": 0.0014367816038429737, "clip_ratio/low_min": 0.0014367816038429737, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0014367816038429737, "reward_total_mean": 0.9942313432693481, "reward_meter_mean": 0.9942313432693481, "reward_meter_std": 0.0012215422466397285, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9942313432693481, "reward_total_composite_std": 0.0012215422466397285} {"timestamp_utc": "2026-04-11T22:41:52Z", "mode": "train", "global_step": 693, "epoch": 0.027834678876973128, "loss": 0.0041, "grad_norm": 10.364151954650879, "learning_rate": 7.903030303030303e-06, "num_tokens": 1507385.0, "completions/mean_length": 54.125, "completions/min_length": 52.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.125, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9447178244590759, "rewards/meter/std": 0.019572317600250244, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9447178244590759, "rewards/total_composite/std": 0.019572317600250244, "reward": 0.9447178244590759, "reward_std": 0.01957232505083084, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024356268346309662, "sampling/sampling_logp_difference/max": 1.9013309478759766, "sampling/importance_sampling_ratio/min": 0.14936968684196472, "sampling/importance_sampling_ratio/mean": 1.003846287727356, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0895584006793797, "clip_ratio/low_mean": 0.009437322150915861, "clip_ratio/low_min": 0.009437322150915861, "clip_ratio/high_mean": 0.01416083937510848, "clip_ratio/high_max": 0.01416083937510848, "clip_ratio/region_mean": 0.02359816152602434, "reward_total_mean": 0.9447178244590759, "reward_meter_mean": 0.9447178244590759, "reward_meter_std": 0.019572317600250244, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9447178244590759, "reward_total_composite_std": 0.019572317600250244} {"timestamp_utc": "2026-04-11T22:41:56Z", "mode": "train", "global_step": 694, "epoch": 0.02787484435875808, "loss": 0.0035, "grad_norm": 6.670380115509033, "learning_rate": 7.9e-06, "num_tokens": 1509273.0, "completions/mean_length": 74.0, "completions/min_length": 70.0, "completions/max_length": 76.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.0, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 76.0, "rewards/meter/mean": 0.610145092010498, "rewards/meter/std": 0.23973627388477325, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.5444035530090332, "rewards/total_composite/std": 0.24921227991580963, "reward": 0.5444035530090332, "reward_std": 0.24921227991580963, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030294643715023994, "sampling/sampling_logp_difference/max": 1.8091429471969604, "sampling/importance_sampling_ratio/min": 0.16379445791244507, "sampling/importance_sampling_ratio/mean": 1.0047415494918823, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.17548873648047447, "clip_ratio/low_mean": 0.01502489356789738, "clip_ratio/low_min": 0.01502489356789738, "clip_ratio/high_mean": 0.02390445303171873, "clip_ratio/high_max": 0.02390445303171873, "clip_ratio/region_mean": 0.03892934659961611, "reward_total_mean": 0.5444035530090332, "reward_meter_mean": 0.610145092010498, "reward_meter_std": 0.23973627388477325, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.17251639068126678, "reward_total_composite_mean": 0.5444035530090332, "reward_total_composite_std": 0.24921227991580963} {"timestamp_utc": "2026-04-11T22:42:01Z", "mode": "train", "global_step": 695, "epoch": 0.02791500984054304, "loss": -0.0124, "grad_norm": 3.7013967037200928, "learning_rate": 7.896969696969698e-06, "num_tokens": 1511199.0, "completions/mean_length": 80.75, "completions/min_length": 78.0, "completions/max_length": 90.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 80.75, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 90.0, "rewards/meter/mean": 0.9964468479156494, "rewards/meter/std": 0.0019429969834163785, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7916666865348816, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.788793683052063, "rewards/total_composite/std": 0.17156179249286652, "reward": 0.788793683052063, "reward_std": 0.17156179249286652, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012971381656825542, "sampling/sampling_logp_difference/max": 1.7286226749420166, "sampling/importance_sampling_ratio/min": 0.1775287538766861, "sampling/importance_sampling_ratio/mean": 1.0024663209915161, "sampling/importance_sampling_ratio/max": 1.8497363328933716, "entropy": 0.06276643788442016, "clip_ratio/low_mean": 0.0062915480230003595, "clip_ratio/low_min": 0.0062915480230003595, "clip_ratio/high_mean": 0.004360056365840137, "clip_ratio/high_max": 0.004360056365840137, "clip_ratio/region_mean": 0.010651604388840497, "reward_total_mean": 0.788793683052063, "reward_meter_mean": 0.9964468479156494, "reward_meter_std": 0.0019429969834163785, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7916666865348816, "reward_repeat_penalty_std": 0.17251639068126678, "reward_total_composite_mean": 0.788793683052063, "reward_total_composite_std": 0.17156179249286652} {"timestamp_utc": "2026-04-11T22:42:06Z", "mode": "train", "global_step": 696, "epoch": 0.027955175322327993, "loss": -0.0063, "grad_norm": 3.119380235671997, "learning_rate": 7.893939393939395e-06, "num_tokens": 1513573.0, "completions/mean_length": 125.75, "completions/min_length": 112.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 125.75, "completions/min_terminated_length": 112.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.9228010177612305, "rewards/meter/std": 0.20124337077140808, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7000000476837158, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.6532999277114868, "rewards/total_composite/std": 0.18960076570510864, "reward": 0.6532999277114868, "reward_std": 0.18960076570510864, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016659488901495934, "sampling/sampling_logp_difference/max": 1.3884687423706055, "sampling/importance_sampling_ratio/min": 0.2494570016860962, "sampling/importance_sampling_ratio/mean": 0.9984103441238403, "sampling/importance_sampling_ratio/max": 1.9363112449645996, "entropy": 0.07676348416134715, "clip_ratio/low_mean": 0.0020850637229159474, "clip_ratio/low_min": 0.0020850637229159474, "clip_ratio/high_mean": 0.00876707280986011, "clip_ratio/high_max": 0.00876707280986011, "clip_ratio/region_mean": 0.010852136532776058, "reward_total_mean": 0.6532999277114868, "reward_meter_mean": 0.9228010177612305, "reward_meter_std": 0.20124337077140808, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7000000476837158, "reward_repeat_penalty_std": 0.10690449178218842, "reward_total_composite_mean": 0.6532999277114868, "reward_total_composite_std": 0.18960076570510864} {"timestamp_utc": "2026-04-11T22:42:11Z", "mode": "train", "global_step": 697, "epoch": 0.027995340804112947, "loss": 0.0606, "grad_norm": 4.277256011962891, "learning_rate": 7.89090909090909e-06, "num_tokens": 1515360.0, "completions/mean_length": 59.375, "completions/min_length": 55.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.375, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9654116630554199, "rewards/meter/std": 0.034304678440093994, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9166666865348816, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.8894380331039429, "rewards/total_composite/std": 0.17405925691127777, "reward": 0.8894380331039429, "reward_std": 0.17405925691127777, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020010532811284065, "sampling/sampling_logp_difference/max": 1.0107874870300293, "sampling/importance_sampling_ratio/min": 0.3639322817325592, "sampling/importance_sampling_ratio/mean": 0.99860018491745, "sampling/importance_sampling_ratio/max": 1.4003076553344727, "entropy": 0.09744885191321373, "clip_ratio/low_mean": 0.003846153849735856, "clip_ratio/low_min": 0.003846153849735856, "clip_ratio/high_mean": 0.01933896285481751, "clip_ratio/high_max": 0.01933896285481751, "clip_ratio/region_mean": 0.023185116704553366, "reward_total_mean": 0.8894380331039429, "reward_meter_mean": 0.9654116630554199, "reward_meter_std": 0.034304678440093994, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9166666865348816, "reward_repeat_penalty_std": 0.15430334210395813, "reward_total_composite_mean": 0.8894380331039429, "reward_total_composite_std": 0.17405925691127777} {"timestamp_utc": "2026-04-11T22:42:21Z", "mode": "train", "global_step": 698, "epoch": 0.0280355062858979, "loss": -0.0873, "grad_norm": 2.2967092990875244, "learning_rate": 7.88787878787879e-06, "num_tokens": 1519450.0, "completions/mean_length": 382.25, "completions/min_length": 321.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 363.71429443359375, "completions/min_terminated_length": 321.0, "completions/max_terminated_length": 412.0, "rewards/meter/mean": 0.6105729937553406, "rewards/meter/std": 0.4790833294391632, "rewards/count_adherence/mean": 0.8854166865348816, "rewards/count_adherence/std": 0.1254950612783432, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/repeat_penalty/mean": 0.3964124917984009, "rewards/repeat_penalty/std": 0.25513792037963867, "rewards/total_composite/mean": 0.18387337028980255, "rewards/total_composite/std": 0.21844197809696198, "reward": 0.18387337028980255, "reward_std": 0.21844197809696198, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0396236851811409, "sampling/sampling_logp_difference/max": 6.693481922149658, "sampling/importance_sampling_ratio/min": 0.001238961354829371, "sampling/importance_sampling_ratio/mean": 0.9987672567367554, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.18515709601342678, "clip_ratio/low_mean": 0.007161757908761501, "clip_ratio/low_min": 0.007161757908761501, "clip_ratio/high_mean": 0.015907755587249994, "clip_ratio/high_max": 0.015907755587249994, "clip_ratio/region_mean": 0.023069513496011496, "reward_total_mean": 0.18387337028980255, "reward_meter_mean": 0.6105729937553406, "reward_meter_std": 0.4790833294391632, "reward_count_adherence_mean": 0.8854166865348816, "reward_count_adherence_std": 0.1254950612783432, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_repeat_penalty_mean": 0.3964124917984009, "reward_repeat_penalty_std": 0.25513792037963867, "reward_total_composite_mean": 0.18387337028980255, "reward_total_composite_std": 0.21844197809696198} {"timestamp_utc": "2026-04-11T22:42:25Z", "mode": "train", "global_step": 699, "epoch": 0.028075671767682855, "loss": 0.0056, "grad_norm": 8.473615646362305, "learning_rate": 7.884848484848485e-06, "num_tokens": 1521331.0, "completions/mean_length": 67.125, "completions/min_length": 63.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.125, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9927859306335449, "rewards/meter/std": 0.014541360549628735, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8333333730697632, "rewards/repeat_penalty/std": 0.17817415297031403, "rewards/total_composite/mean": 0.8282062411308289, "rewards/total_composite/std": 0.18183693289756775, "reward": 0.8282062411308289, "reward_std": 0.18183691799640656, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03374454379081726, "sampling/sampling_logp_difference/max": 1.766160249710083, "sampling/importance_sampling_ratio/min": 0.1795266717672348, "sampling/importance_sampling_ratio/mean": 0.9997567534446716, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.14588068891316652, "clip_ratio/low_mean": 0.00747219193726778, "clip_ratio/low_min": 0.00747219193726778, "clip_ratio/high_mean": 0.02036348171532154, "clip_ratio/high_max": 0.02036348171532154, "clip_ratio/region_mean": 0.02783567365258932, "reward_total_mean": 0.8282062411308289, "reward_meter_mean": 0.9927859306335449, "reward_meter_std": 0.014541360549628735, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8333333730697632, "reward_repeat_penalty_std": 0.17817415297031403, "reward_total_composite_mean": 0.8282062411308289, "reward_total_composite_std": 0.18183693289756775} {"timestamp_utc": "2026-04-11T22:42:35Z", "mode": "train", "global_step": 700, "epoch": 0.02811583724946781, "loss": 0.0097, "grad_norm": 0.8298696279525757, "learning_rate": 7.881818181818182e-06, "num_tokens": 1526242.0, "completions/mean_length": 414.875, "completions/min_length": 401.0, "completions/max_length": 444.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 414.875, "completions/min_terminated_length": 401.0, "completions/max_terminated_length": 444.0, "rewards/meter/mean": 0.9960967302322388, "rewards/meter/std": 0.0013218529056757689, "rewards/count_adherence/mean": 0.84375, "rewards/count_adherence/std": 0.029462797567248344, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.19121241569519043, "rewards/repeat_penalty/std": 0.07494427263736725, "rewards/total_composite/mean": 0.16019713878631592, "rewards/total_composite/std": 0.061176449060440063, "reward": 0.16019713878631592, "reward_std": 0.061176449060440063, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008506695739924908, "sampling/sampling_logp_difference/max": 1.432920217514038, "sampling/importance_sampling_ratio/min": 0.238611102104187, "sampling/importance_sampling_ratio/mean": 1.0012154579162598, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.045434954110533, "clip_ratio/low_mean": 0.00364298140630126, "clip_ratio/low_min": 0.00364298140630126, "clip_ratio/high_mean": 0.006157157360576093, "clip_ratio/high_max": 0.006157157360576093, "clip_ratio/region_mean": 0.009800138766877353, "reward_total_mean": 0.16019713878631592, "reward_meter_mean": 0.9960967302322388, "reward_meter_std": 0.0013218529056757689, "reward_count_adherence_mean": 0.84375, "reward_count_adherence_std": 0.029462797567248344, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.19121241569519043, "reward_repeat_penalty_std": 0.07494427263736725, "reward_total_composite_mean": 0.16019713878631592, "reward_total_composite_std": 0.061176449060440063} {"timestamp_utc": "2026-04-11T22:44:02Z", "mode": "eval", "global_step": 700, "epoch": 0.02811583724946781, "eval_loss": NaN, "eval_runtime": 87.7418, "eval_samples_per_second": 1.185, "eval_steps_per_second": 0.148, "eval_num_tokens": 1526242.0, "eval_completions/mean_length": 249.43269230769232, "eval_completions/min_length": 68.53846153846153, "eval_completions/max_length": 471.9230769230769, "eval_completions/clipped_ratio": 0.11538461538461539, "eval_completions/mean_terminated_length": 217.6739994929387, "eval_completions/min_terminated_length": 68.53846153846153, "eval_completions/max_terminated_length": 415.15384615384613, "eval_rewards/meter/mean": 0.6321157022164419, "eval_rewards/meter/std": 0.37711624113413006, "eval_rewards/count_adherence/mean": 0.8881359283740704, "eval_rewards/count_adherence/std": 0.16248861929545036, "eval_rewards/arabic_clean/mean": 0.9038461538461539, "eval_rewards/arabic_clean/std": 0.2343954168833219, "eval_rewards/repeat_penalty/mean": 0.5953293947073129, "eval_rewards/repeat_penalty/std": 0.32899803152451146, "eval_rewards/total_composite/mean": 0.29438196466519284, "eval_rewards/total_composite/std": 0.28008361991781455, "eval_reward": 0.29438196466519284, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.012457216982371531, "eval_sampling/sampling_logp_difference/max": 0.881988103573139, "eval_sampling/importance_sampling_ratio/min": 0.4343368663237645, "eval_sampling/importance_sampling_ratio/mean": 1.0041690973135142, "eval_sampling/importance_sampling_ratio/max": 1.455959943624643, "eval_entropy": 0.1376804428604933, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.29438196466519284, "eval_reward_meter_mean": 0.6321157022164419, "eval_reward_meter_std": 0.37711624113413006, "eval_reward_count_adherence_mean": 0.8881359283740704, "eval_reward_count_adherence_std": 0.16248861929545036, "eval_reward_arabic_clean_mean": 0.9038461538461539, "eval_reward_arabic_clean_std": 0.2343954168833219, "eval_reward_repeat_penalty_mean": 0.5953293947073129, "eval_reward_repeat_penalty_std": 0.32899803152451146, "eval_reward_total_composite_mean": 0.29438196466519284, "eval_reward_total_composite_std": 0.28008361991781455} {"timestamp_utc": "2026-04-11T22:44:17Z", "mode": "train", "global_step": 701, "epoch": 0.028156002731252763, "loss": -0.12, "grad_norm": 2.5312631130218506, "learning_rate": 7.87878787878788e-06, "num_tokens": 1528168.0, "completions/mean_length": 138.75, "completions/min_length": 79.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 85.42857360839844, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.646858811378479, "rewards/meter/std": 0.40135496854782104, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6387161016464233, "rewards/total_composite/std": 0.41526278853416443, "reward": 0.6387161016464233, "reward_std": 0.41526278853416443, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017966095358133316, "sampling/sampling_logp_difference/max": 0.638648509979248, "sampling/importance_sampling_ratio/min": 0.5280055403709412, "sampling/importance_sampling_ratio/mean": 1.0036530494689941, "sampling/importance_sampling_ratio/max": 1.6849335432052612, "entropy": 0.11331802047789097, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.015958538744598627, "clip_ratio/high_max": 0.015958538744598627, "clip_ratio/region_mean": 0.015958538744598627, "reward_total_mean": 0.6387161016464233, "reward_meter_mean": 0.646858811378479, "reward_meter_std": 0.40135496854782104, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6387161016464233, "reward_total_composite_std": 0.41526278853416443} {"timestamp_utc": "2026-04-11T22:44:27Z", "mode": "train", "global_step": 702, "epoch": 0.028196168213037717, "loss": -0.0691, "grad_norm": 3.0773770809173584, "learning_rate": 7.875757575757577e-06, "num_tokens": 1529770.0, "completions/mean_length": 128.25, "completions/min_length": 71.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 73.42857360839844, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 79.0, "rewards/meter/mean": 0.3803873062133789, "rewards/meter/std": 0.26946189999580383, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.33759811520576477, "rewards/total_composite/std": 0.30163025856018066, "reward": 0.33759811520576477, "reward_std": 0.3016302287578583, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04749147966504097, "sampling/sampling_logp_difference/max": 1.9366450309753418, "sampling/importance_sampling_ratio/min": 0.14418688416481018, "sampling/importance_sampling_ratio/mean": 1.011560082435608, "sampling/importance_sampling_ratio/max": 1.7729606628417969, "entropy": 0.37723080068826675, "clip_ratio/low_mean": 0.01317842910066247, "clip_ratio/low_min": 0.01317842910066247, "clip_ratio/high_mean": 0.0155344782397151, "clip_ratio/high_max": 0.0155344782397151, "clip_ratio/region_mean": 0.02871290734037757, "reward_total_mean": 0.33759811520576477, "reward_meter_mean": 0.3803873062133789, "reward_meter_std": 0.26946189999580383, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.33759811520576477, "reward_total_composite_std": 0.30163025856018066} {"timestamp_utc": "2026-04-11T22:44:34Z", "mode": "train", "global_step": 703, "epoch": 0.02823633369482267, "loss": -0.0245, "grad_norm": 1.2964930534362793, "learning_rate": 7.872727272727273e-06, "num_tokens": 1533320.0, "completions/mean_length": 229.75, "completions/min_length": 206.0, "completions/max_length": 242.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 229.75, "completions/min_terminated_length": 206.0, "completions/max_terminated_length": 242.0, "rewards/meter/mean": 0.9970081448554993, "rewards/meter/std": 0.001187506248243153, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.2613636255264282, "rewards/repeat_penalty/std": 0.197011336684227, "rewards/total_composite/mean": 0.2607119679450989, "rewards/total_composite/std": 0.19680176675319672, "reward": 0.2607119679450989, "reward_std": 0.1968017816543579, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014087834395468235, "sampling/sampling_logp_difference/max": 4.6370849609375, "sampling/importance_sampling_ratio/min": 0.009685891680419445, "sampling/importance_sampling_ratio/mean": 0.9988393187522888, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03166929807048291, "clip_ratio/low_mean": 0.0028346364269964397, "clip_ratio/low_min": 0.0028346364269964397, "clip_ratio/high_mean": 0.004737977171316743, "clip_ratio/high_max": 0.004737977171316743, "clip_ratio/region_mean": 0.007572613598313183, "reward_total_mean": 0.2607119679450989, "reward_meter_mean": 0.9970081448554993, "reward_meter_std": 0.001187506248243153, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.2613636255264282, "reward_repeat_penalty_std": 0.197011336684227, "reward_total_composite_mean": 0.2607119679450989, "reward_total_composite_std": 0.19680176675319672} {"timestamp_utc": "2026-04-11T22:44:44Z", "mode": "train", "global_step": 704, "epoch": 0.028276499176607624, "loss": -0.1074, "grad_norm": 3.015087127685547, "learning_rate": 7.86969696969697e-06, "num_tokens": 1536668.0, "completions/mean_length": 364.5, "completions/min_length": 229.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 315.3333435058594, "completions/min_terminated_length": 229.0, "completions/max_terminated_length": 368.0, "rewards/meter/mean": 0.20065514743328094, "rewards/meter/std": 0.3053584098815918, "rewards/count_adherence/mean": 0.8522727489471436, "rewards/count_adherence/std": 0.21697448194026947, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/repeat_penalty/mean": 0.6299689412117004, "rewards/repeat_penalty/std": 0.28235092759132385, "rewards/total_composite/mean": 0.0969100296497345, "rewards/total_composite/std": 0.21232596039772034, "reward": 0.0969100296497345, "reward_std": 0.21232594549655914, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04316805675625801, "sampling/sampling_logp_difference/max": 2.4648847579956055, "sampling/importance_sampling_ratio/min": 0.08501863479614258, "sampling/importance_sampling_ratio/mean": 0.9969695806503296, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2017015889286995, "clip_ratio/low_mean": 0.01385743310675025, "clip_ratio/low_min": 0.01385743310675025, "clip_ratio/high_mean": 0.009915342554450035, "clip_ratio/high_max": 0.009915342554450035, "clip_ratio/region_mean": 0.023772775661200285, "reward_total_mean": 0.0969100296497345, "reward_meter_mean": 0.20065514743328094, "reward_meter_std": 0.3053584098815918, "reward_count_adherence_mean": 0.8522727489471436, "reward_count_adherence_std": 0.21697448194026947, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_repeat_penalty_mean": 0.6299689412117004, "reward_repeat_penalty_std": 0.28235092759132385, "reward_total_composite_mean": 0.0969100296497345, "reward_total_composite_std": 0.21232596039772034} {"timestamp_utc": "2026-04-11T22:44:49Z", "mode": "train", "global_step": 705, "epoch": 0.02831666465839258, "loss": -0.0075, "grad_norm": 2.615363836288452, "learning_rate": 7.866666666666667e-06, "num_tokens": 1538402.0, "completions/mean_length": 58.75, "completions/min_length": 56.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.75, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9921605587005615, "rewards/meter/std": 0.006017809733748436, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9166666865348816, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.909184455871582, "rewards/total_composite/std": 0.15155236423015594, "reward": 0.909184455871582, "reward_std": 0.15155236423015594, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021948276087641716, "sampling/sampling_logp_difference/max": 3.973776340484619, "sampling/importance_sampling_ratio/min": 0.01880229450762272, "sampling/importance_sampling_ratio/mean": 0.9994084239006042, "sampling/importance_sampling_ratio/max": 1.5749088525772095, "entropy": 0.053199955029413104, "clip_ratio/low_mean": 0.006398809840902686, "clip_ratio/low_min": 0.006398809840902686, "clip_ratio/high_mean": 0.006355932215228677, "clip_ratio/high_max": 0.006355932215228677, "clip_ratio/region_mean": 0.012754742056131363, "reward_total_mean": 0.909184455871582, "reward_meter_mean": 0.9921605587005615, "reward_meter_std": 0.006017809733748436, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9166666865348816, "reward_repeat_penalty_std": 0.15430334210395813, "reward_total_composite_mean": 0.909184455871582, "reward_total_composite_std": 0.15155236423015594} {"timestamp_utc": "2026-04-11T22:44:53Z", "mode": "train", "global_step": 706, "epoch": 0.028356830140177532, "loss": 0.0093, "grad_norm": 4.515374183654785, "learning_rate": 7.863636363636364e-06, "num_tokens": 1539816.0, "completions/mean_length": 44.75, "completions/min_length": 43.0, "completions/max_length": 45.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.75, "completions/min_terminated_length": 43.0, "completions/max_terminated_length": 45.0, "rewards/meter/mean": 0.9157871603965759, "rewards/meter/std": 0.05085242539644241, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9157871603965759, "rewards/total_composite/std": 0.05085242539644241, "reward": 0.9157871603965759, "reward_std": 0.05085243284702301, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02820507250726223, "sampling/sampling_logp_difference/max": 0.7927889823913574, "sampling/importance_sampling_ratio/min": 0.4525808095932007, "sampling/importance_sampling_ratio/mean": 0.997020423412323, "sampling/importance_sampling_ratio/max": 1.2550123929977417, "entropy": 0.13661748263984919, "clip_ratio/low_mean": 0.0027777778450399637, "clip_ratio/low_min": 0.0027777778450399637, "clip_ratio/high_mean": 0.01414728700183332, "clip_ratio/high_max": 0.01414728700183332, "clip_ratio/region_mean": 0.016925064846873283, "reward_total_mean": 0.9157871603965759, "reward_meter_mean": 0.9157871603965759, "reward_meter_std": 0.05085242539644241, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9157871603965759, "reward_total_composite_std": 0.05085242539644241} {"timestamp_utc": "2026-04-11T22:44:59Z", "mode": "train", "global_step": 707, "epoch": 0.028396995621962486, "loss": -0.0003, "grad_norm": 2.2196364402770996, "learning_rate": 7.860606060606062e-06, "num_tokens": 1541901.0, "completions/mean_length": 84.625, "completions/min_length": 83.0, "completions/max_length": 85.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.625, "completions/min_terminated_length": 83.0, "completions/max_terminated_length": 85.0, "rewards/meter/mean": 0.9956607818603516, "rewards/meter/std": 0.0007934165187180042, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9956607818603516, "rewards/total_composite/std": 0.0007934165187180042, "reward": 0.9956607818603516, "reward_std": 0.0007934237364679575, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026080820709466934, "sampling/sampling_logp_difference/max": 0.8360247611999512, "sampling/importance_sampling_ratio/min": 0.4566366374492645, "sampling/importance_sampling_ratio/mean": 1.0091147422790527, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15222040470689535, "clip_ratio/low_mean": 0.007405462441965938, "clip_ratio/low_min": 0.007405462441965938, "clip_ratio/high_mean": 0.013377037481404841, "clip_ratio/high_max": 0.013377037481404841, "clip_ratio/region_mean": 0.02078249992337078, "reward_total_mean": 0.9956607818603516, "reward_meter_mean": 0.9956607818603516, "reward_meter_std": 0.0007934165187180042, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9956607818603516, "reward_total_composite_std": 0.0007934165187180042} {"timestamp_utc": "2026-04-11T22:45:07Z", "mode": "train", "global_step": 708, "epoch": 0.02843716110374744, "loss": 0.0303, "grad_norm": 2.581402063369751, "learning_rate": 7.857575757575759e-06, "num_tokens": 1546099.0, "completions/mean_length": 330.75, "completions/min_length": 284.0, "completions/max_length": 358.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 330.75, "completions/min_terminated_length": 284.0, "completions/max_terminated_length": 358.0, "rewards/meter/mean": 0.5674466490745544, "rewards/meter/std": 0.4153376519680023, "rewards/count_adherence/mean": 0.9027777910232544, "rewards/count_adherence/std": 0.03928370773792267, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.34049707651138306, "rewards/repeat_penalty/std": 0.22640277445316315, "rewards/total_composite/mean": 0.2035118043422699, "rewards/total_composite/std": 0.23429380357265472, "reward": 0.2035118043422699, "reward_std": 0.23429378867149353, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023940538987517357, "sampling/sampling_logp_difference/max": 3.625192880630493, "sampling/importance_sampling_ratio/min": 0.026643957942724228, "sampling/importance_sampling_ratio/mean": 1.002025842666626, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10461874585598707, "clip_ratio/low_mean": 0.012675938894972205, "clip_ratio/low_min": 0.012675938894972205, "clip_ratio/high_mean": 0.010587403550744057, "clip_ratio/high_max": 0.010587403550744057, "clip_ratio/region_mean": 0.023263342445716262, "reward_total_mean": 0.2035118043422699, "reward_meter_mean": 0.5674466490745544, "reward_meter_std": 0.4153376519680023, "reward_count_adherence_mean": 0.9027777910232544, "reward_count_adherence_std": 0.03928370773792267, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.34049707651138306, "reward_repeat_penalty_std": 0.22640277445316315, "reward_total_composite_mean": 0.2035118043422699, "reward_total_composite_std": 0.23429380357265472} {"timestamp_utc": "2026-04-11T22:45:17Z", "mode": "train", "global_step": 709, "epoch": 0.028477326585532394, "loss": -0.3582, "grad_norm": 0.8571996688842773, "learning_rate": 7.854545454545454e-06, "num_tokens": 1549805.0, "completions/mean_length": 452.25, "completions/min_length": 372.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 416.3999938964844, "completions/min_terminated_length": 372.0, "completions/max_terminated_length": 472.0, "rewards/meter/mean": 0.8675932884216309, "rewards/meter/std": 0.35074320435523987, "rewards/count_adherence/mean": 0.6590909361839294, "rewards/count_adherence/std": 0.39101481437683105, "rewards/arabic_clean/mean": 0.625, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/repeat_penalty/mean": 0.6495236158370972, "rewards/repeat_penalty/std": 0.30929210782051086, "rewards/total_composite/mean": 0.25732704997062683, "rewards/total_composite/std": 0.2531158924102783, "reward": 0.25732704997062683, "reward_std": 0.2531158924102783, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019949674606323242, "sampling/sampling_logp_difference/max": 2.8196310997009277, "sampling/importance_sampling_ratio/min": 0.05962793529033661, "sampling/importance_sampling_ratio/mean": 1.0035980939865112, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0702343238517642, "clip_ratio/low_mean": 0.0015743073308840394, "clip_ratio/low_min": 0.0015743073308840394, "clip_ratio/high_mean": 0.005005840037483722, "clip_ratio/high_max": 0.005005840037483722, "clip_ratio/region_mean": 0.006580147368367761, "reward_total_mean": 0.25732704997062683, "reward_meter_mean": 0.8675932884216309, "reward_meter_std": 0.35074320435523987, "reward_count_adherence_mean": 0.6590909361839294, "reward_count_adherence_std": 0.39101481437683105, "reward_arabic_clean_mean": 0.625, "reward_arabic_clean_std": 0.5175492167472839, "reward_repeat_penalty_mean": 0.6495236158370972, "reward_repeat_penalty_std": 0.30929210782051086, "reward_total_composite_mean": 0.25732704997062683, "reward_total_composite_std": 0.2531158924102783} {"timestamp_utc": "2026-04-11T22:45:28Z", "mode": "train", "global_step": 710, "epoch": 0.028517492067317348, "loss": -0.2247, "grad_norm": 0.6216875314712524, "learning_rate": 7.851515151515152e-06, "num_tokens": 1554988.0, "completions/mean_length": 474.875, "completions/min_length": 439.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 469.5714416503906, "completions/min_terminated_length": 439.0, "completions/max_terminated_length": 497.0, "rewards/meter/mean": 0.9897267818450928, "rewards/meter/std": 0.019744135439395905, "rewards/count_adherence/mean": 0.7767857313156128, "rewards/count_adherence/std": 0.20360276103019714, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.5224603414535522, "rewards/repeat_penalty/std": 0.23697951436042786, "rewards/total_composite/mean": 0.3333708643913269, "rewards/total_composite/std": 0.18023891746997833, "reward": 0.3333708643913269, "reward_std": 0.18023891746997833, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010584630072116852, "sampling/sampling_logp_difference/max": 2.4926586151123047, "sampling/importance_sampling_ratio/min": 0.08268983662128448, "sampling/importance_sampling_ratio/mean": 1.0017552375793457, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.05670151812955737, "clip_ratio/low_mean": 0.0021170872496441007, "clip_ratio/low_min": 0.0021170872496441007, "clip_ratio/high_mean": 0.004514876694884151, "clip_ratio/high_max": 0.004514876694884151, "clip_ratio/region_mean": 0.006631963944528252, "reward_total_mean": 0.3333708643913269, "reward_meter_mean": 0.9897267818450928, "reward_meter_std": 0.019744135439395905, "reward_count_adherence_mean": 0.7767857313156128, "reward_count_adherence_std": 0.20360276103019714, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.5224603414535522, "reward_repeat_penalty_std": 0.23697951436042786, "reward_total_composite_mean": 0.3333708643913269, "reward_total_composite_std": 0.18023891746997833} {"timestamp_utc": "2026-04-11T22:45:38Z", "mode": "train", "global_step": 711, "epoch": 0.028557657549102302, "loss": -0.0049, "grad_norm": 0.6046677231788635, "learning_rate": 7.848484848484849e-06, "num_tokens": 1560550.0, "completions/mean_length": 440.25, "completions/min_length": 436.0, "completions/max_length": 461.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 440.25, "completions/min_terminated_length": 436.0, "completions/max_terminated_length": 461.0, "rewards/meter/mean": 0.9979845285415649, "rewards/meter/std": 0.00018585202633403242, "rewards/count_adherence/mean": 0.7946428060531616, "rewards/count_adherence/std": 0.025253823027014732, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.0676877498626709, "rewards/repeat_penalty/std": 0.023803479969501495, "rewards/total_composite/mean": 0.053851306438446045, "rewards/total_composite/std": 0.019493911415338516, "reward": 0.053851306438446045, "reward_std": 0.019493909552693367, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.003577793249860406, "sampling/sampling_logp_difference/max": 1.1886392831802368, "sampling/importance_sampling_ratio/min": 0.30463549494743347, "sampling/importance_sampling_ratio/mean": 1.0001322031021118, "sampling/importance_sampling_ratio/max": 1.460545539855957, "entropy": 0.011412000167183578, "clip_ratio/low_mean": 0.0008561643480788916, "clip_ratio/low_min": 0.0008561643480788916, "clip_ratio/high_mean": 0.0016890882980078459, "clip_ratio/high_max": 0.0016890882980078459, "clip_ratio/region_mean": 0.0025452526460867375, "reward_total_mean": 0.053851306438446045, "reward_meter_mean": 0.9979845285415649, "reward_meter_std": 0.00018585202633403242, "reward_count_adherence_mean": 0.7946428060531616, "reward_count_adherence_std": 0.025253823027014732, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.0676877498626709, "reward_repeat_penalty_std": 0.023803479969501495, "reward_total_composite_mean": 0.053851306438446045, "reward_total_composite_std": 0.019493911415338516} {"timestamp_utc": "2026-04-11T22:45:44Z", "mode": "train", "global_step": 712, "epoch": 0.028597823030887256, "loss": 0.0004, "grad_norm": 4.805282115936279, "learning_rate": 7.845454545454546e-06, "num_tokens": 1562410.0, "completions/mean_length": 69.5, "completions/min_length": 68.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.5, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.8851493000984192, "rewards/meter/std": 0.1646534502506256, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7916666865348816, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.7055873274803162, "rewards/total_composite/std": 0.21653573215007782, "reward": 0.7055873274803162, "reward_std": 0.216535747051239, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03851144388318062, "sampling/sampling_logp_difference/max": 1.9700713157653809, "sampling/importance_sampling_ratio/min": 0.1394468992948532, "sampling/importance_sampling_ratio/mean": 0.997905969619751, "sampling/importance_sampling_ratio/max": 1.8839155435562134, "entropy": 0.1954718241468072, "clip_ratio/low_mean": 0.016306631732732058, "clip_ratio/low_min": 0.016306631732732058, "clip_ratio/high_mean": 0.014235412469133735, "clip_ratio/high_max": 0.014235412469133735, "clip_ratio/region_mean": 0.030542044201865792, "reward_total_mean": 0.7055873274803162, "reward_meter_mean": 0.8851493000984192, "reward_meter_std": 0.1646534502506256, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7916666865348816, "reward_repeat_penalty_std": 0.17251639068126678, "reward_total_composite_mean": 0.7055873274803162, "reward_total_composite_std": 0.21653573215007782} {"timestamp_utc": "2026-04-11T22:45:54Z", "mode": "train", "global_step": 713, "epoch": 0.02863798851267221, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.842424242424243e-06, "num_tokens": 1564234.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.9704335927963257, "rewards/meter/std": 0.07359233498573303, "rewards/count_adherence/mean": 0.8166667222976685, "rewards/count_adherence/std": 0.030860668048262596, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5789903998374939, "rewards/repeat_penalty/std": 0.045423366129398346, "rewards/total_composite/mean": 0.4566306471824646, "rewards/total_composite/std": 0.02297402359545231, "reward": 0.4566306471824646, "reward_std": 0.02297401800751686, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.4566306471824646, "reward_meter_mean": 0.9704335927963257, "reward_meter_std": 0.07359233498573303, "reward_count_adherence_mean": 0.8166667222976685, "reward_count_adherence_std": 0.030860668048262596, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5789903998374939, "reward_repeat_penalty_std": 0.045423366129398346, "reward_total_composite_mean": 0.4566306471824646, "reward_total_composite_std": 0.02297402359545231} {"timestamp_utc": "2026-04-11T22:46:02Z", "mode": "train", "global_step": 714, "epoch": 0.028678153994457164, "loss": 0.0251, "grad_norm": 0.7193350791931152, "learning_rate": 7.83939393939394e-06, "num_tokens": 1568398.0, "completions/mean_length": 336.5, "completions/min_length": 311.0, "completions/max_length": 345.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 336.5, "completions/min_terminated_length": 311.0, "completions/max_terminated_length": 345.0, "rewards/meter/mean": 0.9977849721908569, "rewards/meter/std": 0.0006831432110629976, "rewards/count_adherence/mean": 0.7142857313156128, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.19117647409439087, "rewards/repeat_penalty/std": 0.16262187063694, "rewards/total_composite/mean": 0.13619501888751984, "rewards/total_composite/std": 0.1156935766339302, "reward": 0.13619501888751984, "reward_std": 0.11569356918334961, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.003576630027964711, "sampling/sampling_logp_difference/max": 1.290938377380371, "sampling/importance_sampling_ratio/min": 0.2750125825405121, "sampling/importance_sampling_ratio/mean": 0.9997386932373047, "sampling/importance_sampling_ratio/max": 1.3743659257888794, "entropy": 0.018028545891866088, "clip_ratio/low_mean": 0.0018522579048294574, "clip_ratio/low_min": 0.0018522579048294574, "clip_ratio/high_mean": 0.0012019231216982007, "clip_ratio/high_max": 0.0012019231216982007, "clip_ratio/region_mean": 0.003054181026527658, "reward_total_mean": 0.13619501888751984, "reward_meter_mean": 0.9977849721908569, "reward_meter_std": 0.0006831432110629976, "reward_count_adherence_mean": 0.7142857313156128, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.19117647409439087, "reward_repeat_penalty_std": 0.16262187063694, "reward_total_composite_mean": 0.13619501888751984, "reward_total_composite_std": 0.1156935766339302} {"timestamp_utc": "2026-04-11T22:46:08Z", "mode": "train", "global_step": 715, "epoch": 0.028718319476242118, "loss": 0.0035, "grad_norm": 4.87160062789917, "learning_rate": 7.836363636363638e-06, "num_tokens": 1570787.0, "completions/mean_length": 123.625, "completions/min_length": 115.0, "completions/max_length": 131.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 123.625, "completions/min_terminated_length": 115.0, "completions/max_terminated_length": 131.0, "rewards/meter/mean": 0.9901050329208374, "rewards/meter/std": 0.0037112515419721603, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7321428656578064, "rewards/repeat_penalty/std": 0.1608559489250183, "rewards/total_composite/mean": 0.7246866822242737, "rewards/total_composite/std": 0.15871340036392212, "reward": 0.7246866822242737, "reward_std": 0.15871338546276093, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04571465030312538, "sampling/sampling_logp_difference/max": 1.6292939186096191, "sampling/importance_sampling_ratio/min": 0.19606797397136688, "sampling/importance_sampling_ratio/mean": 0.9998616576194763, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.23239293694496155, "clip_ratio/low_mean": 0.023657660058233887, "clip_ratio/low_min": 0.023657660058233887, "clip_ratio/high_mean": 0.018382353708148003, "clip_ratio/high_max": 0.018382353708148003, "clip_ratio/region_mean": 0.04204001376638189, "reward_total_mean": 0.7246866822242737, "reward_meter_mean": 0.9901050329208374, "reward_meter_std": 0.0037112515419721603, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7321428656578064, "reward_repeat_penalty_std": 0.1608559489250183, "reward_total_composite_mean": 0.7246866822242737, "reward_total_composite_std": 0.15871340036392212} {"timestamp_utc": "2026-04-11T22:46:14Z", "mode": "train", "global_step": 716, "epoch": 0.02875848495802707, "loss": 0.0175, "grad_norm": 2.7540860176086426, "learning_rate": 7.833333333333333e-06, "num_tokens": 1573577.0, "completions/mean_length": 175.75, "completions/min_length": 162.0, "completions/max_length": 215.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 175.75, "completions/min_terminated_length": 162.0, "completions/max_terminated_length": 215.0, "rewards/meter/mean": 0.8235251903533936, "rewards/meter/std": 0.28830382227897644, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6091269850730896, "rewards/repeat_penalty/std": 0.19641855359077454, "rewards/total_composite/mean": 0.4638897478580475, "rewards/total_composite/std": 0.2275657206773758, "reward": 0.4638897478580475, "reward_std": 0.2275657057762146, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02091868966817856, "sampling/sampling_logp_difference/max": 2.359158992767334, "sampling/importance_sampling_ratio/min": 0.09449966251850128, "sampling/importance_sampling_ratio/mean": 1.0048706531524658, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15365072712302208, "clip_ratio/low_mean": 0.010836752247996628, "clip_ratio/low_min": 0.010836752247996628, "clip_ratio/high_mean": 0.0044064579415135086, "clip_ratio/high_max": 0.0044064579415135086, "clip_ratio/region_mean": 0.015243210189510137, "reward_total_mean": 0.4638897478580475, "reward_meter_mean": 0.8235251903533936, "reward_meter_std": 0.28830382227897644, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6091269850730896, "reward_repeat_penalty_std": 0.19641855359077454, "reward_total_composite_mean": 0.4638897478580475, "reward_total_composite_std": 0.2275657206773758} {"timestamp_utc": "2026-04-11T22:46:24Z", "mode": "train", "global_step": 717, "epoch": 0.028798650439812026, "loss": -0.1657, "grad_norm": 1.3285834789276123, "learning_rate": 7.83030303030303e-06, "num_tokens": 1575801.0, "completions/mean_length": 391.0, "completions/min_length": 156.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.625, "completions/mean_terminated_length": 189.33334350585938, "completions/min_terminated_length": 156.0, "completions/max_terminated_length": 210.0, "rewards/meter/mean": 0.3074490427970886, "rewards/meter/std": 0.2025083303451538, "rewards/count_adherence/mean": 0.824999988079071, "rewards/count_adherence/std": 0.12817399203777313, "rewards/arabic_clean/mean": 0.5, "rewards/arabic_clean/std": 0.5345224738121033, "rewards/repeat_penalty/mean": 0.8395833373069763, "rewards/repeat_penalty/std": 0.25571832060813904, "rewards/total_composite/mean": 0.08441510796546936, "rewards/total_composite/std": 0.10946666449308395, "reward": 0.08441510796546936, "reward_std": 0.10946667194366455, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05243195965886116, "sampling/sampling_logp_difference/max": 8.712602615356445, "sampling/importance_sampling_ratio/min": 0.0001644995791139081, "sampling/importance_sampling_ratio/mean": 1.002199649810791, "sampling/importance_sampling_ratio/max": 1.6081585884094238, "entropy": 0.13471947237849236, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.010442280676215887, "clip_ratio/high_max": 0.010442280676215887, "clip_ratio/region_mean": 0.010442280676215887, "reward_total_mean": 0.08441510796546936, "reward_meter_mean": 0.3074490427970886, "reward_meter_std": 0.2025083303451538, "reward_count_adherence_mean": 0.824999988079071, "reward_count_adherence_std": 0.12817399203777313, "reward_arabic_clean_mean": 0.5, "reward_arabic_clean_std": 0.5345224738121033, "reward_repeat_penalty_mean": 0.8395833373069763, "reward_repeat_penalty_std": 0.25571832060813904, "reward_total_composite_mean": 0.08441510796546936, "reward_total_composite_std": 0.10946666449308395} {"timestamp_utc": "2026-04-11T22:46:29Z", "mode": "train", "global_step": 718, "epoch": 0.02883881592159698, "loss": 0.0058, "grad_norm": 6.637593746185303, "learning_rate": 7.827272727272728e-06, "num_tokens": 1577519.0, "completions/mean_length": 57.75, "completions/min_length": 57.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.75, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9619162082672119, "rewards/meter/std": 0.06612562388181686, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9619162082672119, "rewards/total_composite/std": 0.06612562388181686, "reward": 0.9619162082672119, "reward_std": 0.06612562388181686, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03136257454752922, "sampling/sampling_logp_difference/max": 1.5830154418945312, "sampling/importance_sampling_ratio/min": 0.20535492897033691, "sampling/importance_sampling_ratio/mean": 1.004167914390564, "sampling/importance_sampling_ratio/max": 1.769487977027893, "entropy": 0.14472659677267075, "clip_ratio/low_mean": 0.006465517450124025, "clip_ratio/low_min": 0.006465517450124025, "clip_ratio/high_mean": 0.023710741428658366, "clip_ratio/high_max": 0.023710741428658366, "clip_ratio/region_mean": 0.03017625887878239, "reward_total_mean": 0.9619162082672119, "reward_meter_mean": 0.9619162082672119, "reward_meter_std": 0.06612562388181686, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9619162082672119, "reward_total_composite_std": 0.06612562388181686} {"timestamp_utc": "2026-04-11T22:46:39Z", "mode": "train", "global_step": 719, "epoch": 0.028878981403381934, "loss": -0.1972, "grad_norm": 0.982711136341095, "learning_rate": 7.824242424242425e-06, "num_tokens": 1580014.0, "completions/mean_length": 208.875, "completions/min_length": 142.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 165.57144165039062, "completions/min_terminated_length": 142.0, "completions/max_terminated_length": 171.0, "rewards/meter/mean": 0.8615807294845581, "rewards/meter/std": 0.1756223738193512, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.5022321939468384, "rewards/repeat_penalty/std": 0.19195686280727386, "rewards/total_composite/mean": 0.36472243070602417, "rewards/total_composite/std": 0.1904788464307785, "reward": 0.36472243070602417, "reward_std": 0.1904788315296173, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011406843550503254, "sampling/sampling_logp_difference/max": 0.5394062995910645, "sampling/importance_sampling_ratio/min": 0.5830943584442139, "sampling/importance_sampling_ratio/mean": 1.0052262544631958, "sampling/importance_sampling_ratio/max": 1.6336287260055542, "entropy": 0.07858177460730076, "clip_ratio/low_mean": 0.0008802816737443209, "clip_ratio/low_min": 0.0008802816737443209, "clip_ratio/high_mean": 0.005198180675506592, "clip_ratio/high_max": 0.005198180675506592, "clip_ratio/region_mean": 0.006078462349250913, "reward_total_mean": 0.36472243070602417, "reward_meter_mean": 0.8615807294845581, "reward_meter_std": 0.1756223738193512, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.5022321939468384, "reward_repeat_penalty_std": 0.19195686280727386, "reward_total_composite_mean": 0.36472243070602417, "reward_total_composite_std": 0.1904788464307785} {"timestamp_utc": "2026-04-11T22:46:47Z", "mode": "train", "global_step": 720, "epoch": 0.028919146885166887, "loss": -0.0478, "grad_norm": 1.5212147235870361, "learning_rate": 7.821212121212122e-06, "num_tokens": 1583419.0, "completions/mean_length": 228.625, "completions/min_length": 195.0, "completions/max_length": 243.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 228.625, "completions/min_terminated_length": 195.0, "completions/max_terminated_length": 243.0, "rewards/meter/mean": 0.9813181161880493, "rewards/meter/std": 0.009512215852737427, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.06613000482320786, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.527634859085083, "rewards/repeat_penalty/std": 0.18670302629470825, "rewards/total_composite/mean": 0.3818710744380951, "rewards/total_composite/std": 0.13756081461906433, "reward": 0.3818710744380951, "reward_std": 0.13756079971790314, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014808963052928448, "sampling/sampling_logp_difference/max": 1.9443225860595703, "sampling/importance_sampling_ratio/min": 0.14308412373065948, "sampling/importance_sampling_ratio/mean": 1.0003083944320679, "sampling/importance_sampling_ratio/max": 1.7837820053100586, "entropy": 0.08757321583107114, "clip_ratio/low_mean": 0.0012820513220503926, "clip_ratio/low_min": 0.0012820513220503926, "clip_ratio/high_mean": 0.015587894711643457, "clip_ratio/high_max": 0.015587894711643457, "clip_ratio/region_mean": 0.01686994603369385, "reward_total_mean": 0.3818710744380951, "reward_meter_mean": 0.9813181161880493, "reward_meter_std": 0.009512215852737427, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.06613000482320786, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.527634859085083, "reward_repeat_penalty_std": 0.18670302629470825, "reward_total_composite_mean": 0.3818710744380951, "reward_total_composite_std": 0.13756081461906433} {"timestamp_utc": "2026-04-11T22:46:52Z", "mode": "train", "global_step": 721, "epoch": 0.02895931236695184, "loss": -0.0114, "grad_norm": 9.25263786315918, "learning_rate": 7.81818181818182e-06, "num_tokens": 1585267.0, "completions/mean_length": 65.0, "completions/min_length": 60.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.0, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.6475868821144104, "rewards/meter/std": 0.3421926498413086, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9166666865348816, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.5946333408355713, "rewards/total_composite/std": 0.34638410806655884, "reward": 0.5946333408355713, "reward_std": 0.34638410806655884, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1203828752040863, "sampling/sampling_logp_difference/max": 2.941412925720215, "sampling/importance_sampling_ratio/min": 0.052791085094213486, "sampling/importance_sampling_ratio/mean": 0.991131603717804, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.5916384495794773, "clip_ratio/low_mean": 0.051155281253159046, "clip_ratio/low_min": 0.051155281253159046, "clip_ratio/high_mean": 0.05381742771714926, "clip_ratio/high_max": 0.05381742771714926, "clip_ratio/region_mean": 0.1049727089703083, "reward_total_mean": 0.5946333408355713, "reward_meter_mean": 0.6475868821144104, "reward_meter_std": 0.3421926498413086, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9166666865348816, "reward_repeat_penalty_std": 0.15430334210395813, "reward_total_composite_mean": 0.5946333408355713, "reward_total_composite_std": 0.34638410806655884} {"timestamp_utc": "2026-04-11T22:46:57Z", "mode": "train", "global_step": 722, "epoch": 0.028999477848736795, "loss": 0.3273, "grad_norm": 7.886468410491943, "learning_rate": 7.815151515151515e-06, "num_tokens": 1586837.0, "completions/mean_length": 48.25, "completions/min_length": 42.0, "completions/max_length": 92.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.25, "completions/min_terminated_length": 42.0, "completions/max_terminated_length": 92.0, "rewards/meter/mean": 0.9944906830787659, "rewards/meter/std": 0.0011895333882421255, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8698122501373291, "rewards/total_composite/std": 0.35145723819732666, "reward": 0.8698122501373291, "reward_std": 0.35145723819732666, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01547156646847725, "sampling/sampling_logp_difference/max": 0.7411696910858154, "sampling/importance_sampling_ratio/min": 0.4765561819076538, "sampling/importance_sampling_ratio/mean": 1.0014286041259766, "sampling/importance_sampling_ratio/max": 1.3288551568984985, "entropy": 0.10689277853816748, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8698122501373291, "reward_meter_mean": 0.9944906830787659, "reward_meter_std": 0.0011895333882421255, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8698122501373291, "reward_total_composite_std": 0.35145723819732666} {"timestamp_utc": "2026-04-11T22:47:08Z", "mode": "train", "global_step": 723, "epoch": 0.02903964333052175, "loss": -0.2014, "grad_norm": 0.8348656892776489, "learning_rate": 7.812121212121213e-06, "num_tokens": 1590899.0, "completions/mean_length": 485.75, "completions/min_length": 452.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 470.0, "completions/min_terminated_length": 452.0, "completions/max_terminated_length": 494.0, "rewards/meter/mean": 0.996362566947937, "rewards/meter/std": 0.0027744658291339874, "rewards/count_adherence/mean": 0.9464285373687744, "rewards/count_adherence/std": 0.03306501731276512, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.10470085591077805, "rewards/repeat_penalty/std": 0.10408195108175278, "rewards/total_composite/mean": 0.09887628257274628, "rewards/total_composite/std": 0.09698235988616943, "reward": 0.09887628257274628, "reward_std": 0.09698235988616943, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007093369495123625, "sampling/sampling_logp_difference/max": 1.990865707397461, "sampling/importance_sampling_ratio/min": 0.13657712936401367, "sampling/importance_sampling_ratio/mean": 1.0003700256347656, "sampling/importance_sampling_ratio/max": 1.536428451538086, "entropy": 0.018016068963333964, "clip_ratio/low_mean": 0.0007961043156683445, "clip_ratio/low_min": 0.0007961043156683445, "clip_ratio/high_mean": 0.0007826215587556362, "clip_ratio/high_max": 0.0007826215587556362, "clip_ratio/region_mean": 0.0015787258744239807, "reward_total_mean": 0.09887628257274628, "reward_meter_mean": 0.996362566947937, "reward_meter_std": 0.0027744658291339874, "reward_count_adherence_mean": 0.9464285373687744, "reward_count_adherence_std": 0.03306501731276512, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.10470085591077805, "reward_repeat_penalty_std": 0.10408195108175278, "reward_total_composite_mean": 0.09887628257274628, "reward_total_composite_std": 0.09698235988616943} {"timestamp_utc": "2026-04-11T22:47:17Z", "mode": "train", "global_step": 724, "epoch": 0.029079808812306703, "loss": 0.0117, "grad_norm": 1.2350369691848755, "learning_rate": 7.80909090909091e-06, "num_tokens": 1595768.0, "completions/mean_length": 390.625, "completions/min_length": 379.0, "completions/max_length": 403.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 390.625, "completions/min_terminated_length": 379.0, "completions/max_terminated_length": 403.0, "rewards/meter/mean": 0.9978216886520386, "rewards/meter/std": 0.0007184247369877994, "rewards/count_adherence/mean": 0.734375, "rewards/count_adherence/std": 0.04419417306780815, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.17863407731056213, "rewards/repeat_penalty/std": 0.16910506784915924, "rewards/total_composite/mean": 0.1255086362361908, "rewards/total_composite/std": 0.10828039050102234, "reward": 0.1255086362361908, "reward_std": 0.10828038305044174, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0057006035931408405, "sampling/sampling_logp_difference/max": 1.5134329795837402, "sampling/importance_sampling_ratio/min": 0.22015291452407837, "sampling/importance_sampling_ratio/mean": 0.9993754625320435, "sampling/importance_sampling_ratio/max": 1.8996793031692505, "entropy": 0.028950548847205937, "clip_ratio/low_mean": 0.0028556776233017445, "clip_ratio/low_min": 0.0028556776233017445, "clip_ratio/high_mean": 0.002920488826930523, "clip_ratio/high_max": 0.002920488826930523, "clip_ratio/region_mean": 0.005776166450232267, "reward_total_mean": 0.1255086362361908, "reward_meter_mean": 0.9978216886520386, "reward_meter_std": 0.0007184247369877994, "reward_count_adherence_mean": 0.734375, "reward_count_adherence_std": 0.04419417306780815, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.17863407731056213, "reward_repeat_penalty_std": 0.16910506784915924, "reward_total_composite_mean": 0.1255086362361908, "reward_total_composite_std": 0.10828039050102234} {"timestamp_utc": "2026-04-11T22:47:26Z", "mode": "train", "global_step": 725, "epoch": 0.029119974294091657, "loss": -0.1213, "grad_norm": 1.6505030393600464, "learning_rate": 7.806060606060607e-06, "num_tokens": 1597686.0, "completions/mean_length": 136.75, "completions/min_length": 78.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 83.14286041259766, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.5868990421295166, "rewards/meter/std": 0.38611555099487305, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.5530991554260254, "rewards/total_composite/std": 0.40852445363998413, "reward": 0.5530991554260254, "reward_std": 0.4085244834423065, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022672384977340698, "sampling/sampling_logp_difference/max": 0.911320686340332, "sampling/importance_sampling_ratio/min": 0.4019929766654968, "sampling/importance_sampling_ratio/mean": 1.0019892454147339, "sampling/importance_sampling_ratio/max": 1.7016674280166626, "entropy": 0.09931796230375767, "clip_ratio/low_mean": 0.004787406767718494, "clip_ratio/low_min": 0.004787406767718494, "clip_ratio/high_mean": 0.007217321544885635, "clip_ratio/high_max": 0.007217321544885635, "clip_ratio/region_mean": 0.01200472831260413, "reward_total_mean": 0.5530991554260254, "reward_meter_mean": 0.5868990421295166, "reward_meter_std": 0.38611555099487305, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.5530991554260254, "reward_total_composite_std": 0.40852445363998413} {"timestamp_utc": "2026-04-11T22:47:32Z", "mode": "train", "global_step": 726, "epoch": 0.02916013977587661, "loss": -0.0023, "grad_norm": 1.0429009199142456, "learning_rate": 7.803030303030303e-06, "num_tokens": 1600008.0, "completions/mean_length": 127.25, "completions/min_length": 126.0, "completions/max_length": 129.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 127.25, "completions/min_terminated_length": 126.0, "completions/max_terminated_length": 129.0, "rewards/meter/mean": 0.9302235841751099, "rewards/meter/std": 0.015423719771206379, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5, "rewards/repeat_penalty/std": 0.1511857956647873, "rewards/total_composite/mean": 0.46501004695892334, "rewards/total_composite/std": 0.14159858226776123, "reward": 0.46501004695892334, "reward_std": 0.14159856736660004, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01078812312334776, "sampling/sampling_logp_difference/max": 0.9157018661499023, "sampling/importance_sampling_ratio/min": 0.4002356231212616, "sampling/importance_sampling_ratio/mean": 1.002754807472229, "sampling/importance_sampling_ratio/max": 1.5183614492416382, "entropy": 0.05762754753232002, "clip_ratio/low_mean": 0.0029605674790218472, "clip_ratio/low_min": 0.0029605674790218472, "clip_ratio/high_mean": 0.0009765625, "clip_ratio/high_max": 0.0009765625, "clip_ratio/region_mean": 0.003937129979021847, "reward_total_mean": 0.46501004695892334, "reward_meter_mean": 0.9302235841751099, "reward_meter_std": 0.015423719771206379, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5, "reward_repeat_penalty_std": 0.1511857956647873, "reward_total_composite_mean": 0.46501004695892334, "reward_total_composite_std": 0.14159858226776123} {"timestamp_utc": "2026-04-11T22:47:37Z", "mode": "train", "global_step": 727, "epoch": 0.029200305257661565, "loss": 0.0279, "grad_norm": 7.388904094696045, "learning_rate": 7.800000000000002e-06, "num_tokens": 1601923.0, "completions/mean_length": 99.375, "completions/min_length": 93.0, "completions/max_length": 104.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 99.375, "completions/min_terminated_length": 93.0, "completions/max_terminated_length": 104.0, "rewards/meter/mean": 0.4179952144622803, "rewards/meter/std": 0.45823782682418823, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.3363860249519348, "rewards/total_composite/std": 0.36587125062942505, "reward": 0.3363860249519348, "reward_std": 0.36587125062942505, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0847281664609909, "sampling/sampling_logp_difference/max": 5.028669357299805, "sampling/importance_sampling_ratio/min": 0.006547517143189907, "sampling/importance_sampling_ratio/mean": 0.9850558042526245, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2684608269482851, "clip_ratio/low_mean": 0.032587712397798896, "clip_ratio/low_min": 0.032587712397798896, "clip_ratio/high_mean": 0.02533797360956669, "clip_ratio/high_max": 0.02533797360956669, "clip_ratio/region_mean": 0.057925686007365584, "reward_total_mean": 0.3363860249519348, "reward_meter_mean": 0.4179952144622803, "reward_meter_std": 0.45823782682418823, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.10690449178218842, "reward_total_composite_mean": 0.3363860249519348, "reward_total_composite_std": 0.36587125062942505} {"timestamp_utc": "2026-04-11T22:47:46Z", "mode": "train", "global_step": 728, "epoch": 0.02924047073944652, "loss": 0.0076, "grad_norm": 2.0805506706237793, "learning_rate": 7.796969696969697e-06, "num_tokens": 1607332.0, "completions/mean_length": 459.125, "completions/min_length": 427.0, "completions/max_length": 483.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 459.125, "completions/min_terminated_length": 427.0, "completions/max_terminated_length": 483.0, "rewards/meter/mean": 0.28689688444137573, "rewards/meter/std": 0.3819184899330139, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.0431290864944458, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5729086995124817, "rewards/repeat_penalty/std": 0.018385794013738632, "rewards/total_composite/mean": 0.12284211814403534, "rewards/total_composite/std": 0.1601785272359848, "reward": 0.12284211814403534, "reward_std": 0.1601785272359848, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019973881542682648, "sampling/sampling_logp_difference/max": 3.37805438041687, "sampling/importance_sampling_ratio/min": 0.03411376476287842, "sampling/importance_sampling_ratio/mean": 0.9993253350257874, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06644301256164908, "clip_ratio/low_mean": 0.008432107453700155, "clip_ratio/low_min": 0.008432107453700155, "clip_ratio/high_mean": 0.003258531214669347, "clip_ratio/high_max": 0.003258531214669347, "clip_ratio/region_mean": 0.011690638668369502, "reward_total_mean": 0.12284211814403534, "reward_meter_mean": 0.28689688444137573, "reward_meter_std": 0.3819184899330139, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.0431290864944458, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5729086995124817, "reward_repeat_penalty_std": 0.018385794013738632, "reward_total_composite_mean": 0.12284211814403534, "reward_total_composite_std": 0.1601785272359848} {"timestamp_utc": "2026-04-11T22:47:56Z", "mode": "train", "global_step": 729, "epoch": 0.029280636221231473, "loss": -0.1369, "grad_norm": 2.659188747406006, "learning_rate": 7.793939393939394e-06, "num_tokens": 1609182.0, "completions/mean_length": 139.25, "completions/min_length": 80.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 86.00000762939453, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.700344443321228, "rewards/meter/std": 0.42837557196617126, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6998236179351807, "rewards/total_composite/std": 0.4293443560600281, "reward": 0.6998236179351807, "reward_std": 0.4293443262577057, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022995855659246445, "sampling/sampling_logp_difference/max": 0.6324782371520996, "sampling/importance_sampling_ratio/min": 0.5312735438346863, "sampling/importance_sampling_ratio/mean": 1.0047773122787476, "sampling/importance_sampling_ratio/max": 1.5019832849502563, "entropy": 0.1982464650645852, "clip_ratio/low_mean": 0.0015625000232830644, "clip_ratio/low_min": 0.0015625000232830644, "clip_ratio/high_mean": 0.007218867307528853, "clip_ratio/high_max": 0.007218867307528853, "clip_ratio/region_mean": 0.008781367330811918, "reward_total_mean": 0.6998236179351807, "reward_meter_mean": 0.700344443321228, "reward_meter_std": 0.42837557196617126, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6998236179351807, "reward_total_composite_std": 0.4293443560600281} {"timestamp_utc": "2026-04-11T22:48:01Z", "mode": "train", "global_step": 730, "epoch": 0.029320801703016427, "loss": 0.0222, "grad_norm": 3.446493625640869, "learning_rate": 7.790909090909092e-06, "num_tokens": 1611175.0, "completions/mean_length": 84.125, "completions/min_length": 78.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 84.125, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.9330159425735474, "rewards/meter/std": 0.04698435589671135, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9330159425735474, "rewards/total_composite/std": 0.04698435589671135, "reward": 0.9330159425735474, "reward_std": 0.04698435962200165, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02119125984609127, "sampling/sampling_logp_difference/max": 1.4188241958618164, "sampling/importance_sampling_ratio/min": 0.3314521312713623, "sampling/importance_sampling_ratio/mean": 1.0041048526763916, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.13521220535039902, "clip_ratio/low_mean": 0.002890269970521331, "clip_ratio/low_min": 0.002890269970521331, "clip_ratio/high_mean": 0.010773762594908476, "clip_ratio/high_max": 0.010773762594908476, "clip_ratio/region_mean": 0.013664032565429807, "reward_total_mean": 0.9330159425735474, "reward_meter_mean": 0.9330159425735474, "reward_meter_std": 0.04698435589671135, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9330159425735474, "reward_total_composite_std": 0.04698435589671135} {"timestamp_utc": "2026-04-11T22:48:11Z", "mode": "train", "global_step": 731, "epoch": 0.02936096718480138, "loss": -0.2761, "grad_norm": 3.5135695934295654, "learning_rate": 7.787878787878789e-06, "num_tokens": 1613214.0, "completions/mean_length": 506.875, "completions/min_length": 471.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.875, "completions/mean_terminated_length": 471.0, "completions/min_terminated_length": 471.0, "completions/max_terminated_length": 471.0, "rewards/meter/mean": 0.9958064556121826, "rewards/meter/std": 0.0007912082364782691, "rewards/count_adherence/mean": 0.5277777910232544, "rewards/count_adherence/std": 0.051434461027383804, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.11126373708248138, "rewards/repeat_penalty/std": 0.07416856288909912, "rewards/total_composite/mean": 0.055620092898607254, "rewards/total_composite/std": 0.029501251876354218, "reward": 0.055620092898607254, "reward_std": 0.02950124815106392, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013655466958880424, "sampling/sampling_logp_difference/max": 1.3458614349365234, "sampling/importance_sampling_ratio/min": 0.2603153586387634, "sampling/importance_sampling_ratio/mean": 0.9985577464103699, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.005113576073199511, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0005307855899445713, "clip_ratio/high_max": 0.0005307855899445713, "clip_ratio/region_mean": 0.0005307855899445713, "reward_total_mean": 0.055620092898607254, "reward_meter_mean": 0.9958064556121826, "reward_meter_std": 0.0007912082364782691, "reward_count_adherence_mean": 0.5277777910232544, "reward_count_adherence_std": 0.051434461027383804, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.11126373708248138, "reward_repeat_penalty_std": 0.07416856288909912, "reward_total_composite_mean": 0.055620092898607254, "reward_total_composite_std": 0.029501251876354218} {"timestamp_utc": "2026-04-11T22:48:21Z", "mode": "train", "global_step": 732, "epoch": 0.029401132666586335, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.784848484848484e-06, "num_tokens": 1614982.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.9962900280952454, "rewards/meter/std": 0.001235451316460967, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0345032773911953, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.19743433594703674, "rewards/repeat_penalty/std": 0.17679214477539062, "rewards/total_composite/mean": 0.18774111568927765, "rewards/total_composite/std": 0.16288606822490692, "reward": 0.18774111568927765, "reward_std": 0.16288605332374573, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.18774111568927765, "reward_meter_mean": 0.9962900280952454, "reward_meter_std": 0.001235451316460967, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0345032773911953, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.19743433594703674, "reward_repeat_penalty_std": 0.17679214477539062, "reward_total_composite_mean": 0.18774111568927765, "reward_total_composite_std": 0.16288606822490692} {"timestamp_utc": "2026-04-11T22:48:26Z", "mode": "train", "global_step": 733, "epoch": 0.02944129814837129, "loss": 0.026, "grad_norm": 5.411409378051758, "learning_rate": 7.781818181818183e-06, "num_tokens": 1616806.0, "completions/mean_length": 72.0, "completions/min_length": 63.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.0, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.6948316097259521, "rewards/meter/std": 0.2794814109802246, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8333333730697632, "rewards/repeat_penalty/std": 0.17817415297031403, "rewards/total_composite/mean": 0.5835638046264648, "rewards/total_composite/std": 0.28273841738700867, "reward": 0.5835638046264648, "reward_std": 0.28273844718933105, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.055757418274879456, "sampling/sampling_logp_difference/max": 3.232696771621704, "sampling/importance_sampling_ratio/min": 0.03945096582174301, "sampling/importance_sampling_ratio/mean": 1.0083332061767578, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.3269932046532631, "clip_ratio/low_mean": 0.034895967692136765, "clip_ratio/low_min": 0.034895967692136765, "clip_ratio/high_mean": 0.027756539173424244, "clip_ratio/high_max": 0.027756539173424244, "clip_ratio/region_mean": 0.06265250686556101, "reward_total_mean": 0.5835638046264648, "reward_meter_mean": 0.6948316097259521, "reward_meter_std": 0.2794814109802246, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8333333730697632, "reward_repeat_penalty_std": 0.17817415297031403, "reward_total_composite_mean": 0.5835638046264648, "reward_total_composite_std": 0.28273841738700867} {"timestamp_utc": "2026-04-11T22:48:31Z", "mode": "train", "global_step": 734, "epoch": 0.029481463630156243, "loss": 0.0012, "grad_norm": 1.1426414251327515, "learning_rate": 7.778787878787879e-06, "num_tokens": 1619174.0, "completions/mean_length": 116.0, "completions/min_length": 116.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 116.0, "completions/min_terminated_length": 116.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.9807324409484863, "rewards/meter/std": 0.005540105979889631, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8406277894973755, "rewards/total_composite/std": 0.004748670384287834, "reward": 0.8406277894973755, "reward_std": 0.004748655948787928, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004732904955744743, "sampling/sampling_logp_difference/max": 0.8726806640625, "sampling/importance_sampling_ratio/min": 0.4178299903869629, "sampling/importance_sampling_ratio/mean": 1.0012872219085693, "sampling/importance_sampling_ratio/max": 1.4202854633331299, "entropy": 0.025226276833564043, "clip_ratio/low_mean": 0.0021551724057644606, "clip_ratio/low_min": 0.0021551724057644606, "clip_ratio/high_mean": 0.0010775862028822303, "clip_ratio/high_max": 0.0010775862028822303, "clip_ratio/region_mean": 0.003232758608646691, "reward_total_mean": 0.8406277894973755, "reward_meter_mean": 0.9807324409484863, "reward_meter_std": 0.005540105979889631, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8406277894973755, "reward_total_composite_std": 0.004748670384287834} {"timestamp_utc": "2026-04-11T22:48:36Z", "mode": "train", "global_step": 735, "epoch": 0.029521629111941197, "loss": 0.0021, "grad_norm": 6.164267539978027, "learning_rate": 7.775757575757576e-06, "num_tokens": 1620938.0, "completions/mean_length": 77.5, "completions/min_length": 74.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.5, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.8432132005691528, "rewards/meter/std": 0.11800947040319443, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9249999523162842, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.7734056115150452, "rewards/total_composite/std": 0.09497449547052383, "reward": 0.7734056115150452, "reward_std": 0.09497448056936264, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04740333557128906, "sampling/sampling_logp_difference/max": 1.3389692306518555, "sampling/importance_sampling_ratio/min": 0.2621157169342041, "sampling/importance_sampling_ratio/mean": 1.0057860612869263, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2497086301445961, "clip_ratio/low_mean": 0.026082158088684082, "clip_ratio/low_min": 0.026082158088684082, "clip_ratio/high_mean": 0.011470985249616206, "clip_ratio/high_max": 0.011470985249616206, "clip_ratio/region_mean": 0.03755314333830029, "reward_total_mean": 0.7734056115150452, "reward_meter_mean": 0.8432132005691528, "reward_meter_std": 0.11800947040319443, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9249999523162842, "reward_repeat_penalty_std": 0.1035098284482956, "reward_total_composite_mean": 0.7734056115150452, "reward_total_composite_std": 0.09497449547052383} {"timestamp_utc": "2026-04-11T22:48:44Z", "mode": "train", "global_step": 736, "epoch": 0.02956179459372615, "loss": 0.46, "grad_norm": 5.787602424621582, "learning_rate": 7.772727272727273e-06, "num_tokens": 1623113.0, "completions/mean_length": 115.875, "completions/min_length": 72.0, "completions/max_length": 353.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 115.875, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 353.0, "rewards/meter/mean": 0.7237052917480469, "rewards/meter/std": 0.3117098808288574, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6369646191596985, "rewards/total_composite/std": 0.40405309200286865, "reward": 0.6369646191596985, "reward_std": 0.40405309200286865, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.08920754492282867, "sampling/sampling_logp_difference/max": 1.3201618194580078, "sampling/importance_sampling_ratio/min": 0.2670920789241791, "sampling/importance_sampling_ratio/mean": 1.0180284976959229, "sampling/importance_sampling_ratio/max": 1.964148998260498, "entropy": 1.2863622568547726, "clip_ratio/low_mean": 0.008429729146882892, "clip_ratio/low_min": 0.008429729146882892, "clip_ratio/high_mean": 0.01677176496013999, "clip_ratio/high_max": 0.01677176496013999, "clip_ratio/region_mean": 0.02520149410702288, "reward_total_mean": 0.6369646191596985, "reward_meter_mean": 0.7237052917480469, "reward_meter_std": 0.3117098808288574, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6369646191596985, "reward_total_composite_std": 0.40405309200286865} {"timestamp_utc": "2026-04-11T22:48:54Z", "mode": "train", "global_step": 737, "epoch": 0.029601960075511104, "loss": -0.2132, "grad_norm": 1.1047923564910889, "learning_rate": 7.76969696969697e-06, "num_tokens": 1625144.0, "completions/mean_length": 294.875, "completions/min_length": 147.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.375, "completions/mean_terminated_length": 164.60000610351562, "completions/min_terminated_length": 147.0, "completions/max_terminated_length": 191.0, "rewards/meter/mean": 0.3803099989891052, "rewards/meter/std": 0.2973701059818268, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.22160132229328156, "rewards/arabic_clean/mean": 0.625, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/repeat_penalty/mean": 0.797619104385376, "rewards/repeat_penalty/std": 0.2185886949300766, "rewards/total_composite/mean": 0.20652014017105103, "rewards/total_composite/std": 0.19584397971630096, "reward": 0.20652014017105103, "reward_std": 0.19584397971630096, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030263110995292664, "sampling/sampling_logp_difference/max": 1.4690535068511963, "sampling/importance_sampling_ratio/min": 0.2301432192325592, "sampling/importance_sampling_ratio/mean": 1.0015567541122437, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.12422919739037752, "clip_ratio/low_mean": 0.0008333333535119891, "clip_ratio/low_min": 0.0008333333535119891, "clip_ratio/high_mean": 0.015350477071478963, "clip_ratio/high_max": 0.015350477071478963, "clip_ratio/region_mean": 0.016183810424990952, "reward_total_mean": 0.20652014017105103, "reward_meter_mean": 0.3803099989891052, "reward_meter_std": 0.2973701059818268, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.22160132229328156, "reward_arabic_clean_mean": 0.625, "reward_arabic_clean_std": 0.5175492167472839, "reward_repeat_penalty_mean": 0.797619104385376, "reward_repeat_penalty_std": 0.2185886949300766, "reward_total_composite_mean": 0.20652014017105103, "reward_total_composite_std": 0.19584397971630096} {"timestamp_utc": "2026-04-11T22:49:04Z", "mode": "train", "global_step": 738, "epoch": 0.02964212555729606, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.766666666666666e-06, "num_tokens": 1626768.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.9730854630470276, "rewards/meter/std": 0.03588658943772316, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.050507619976997375, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.2781907618045807, "rewards/repeat_penalty/std": 0.2387264519929886, "rewards/total_composite/mean": 0.2381971776485443, "rewards/total_composite/std": 0.20977550745010376, "reward": 0.2381971776485443, "reward_std": 0.20977550745010376, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.2381971776485443, "reward_meter_mean": 0.9730854630470276, "reward_meter_std": 0.03588658943772316, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.050507619976997375, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.2781907618045807, "reward_repeat_penalty_std": 0.2387264519929886, "reward_total_composite_mean": 0.2381971776485443, "reward_total_composite_std": 0.20977550745010376} {"timestamp_utc": "2026-04-11T22:49:13Z", "mode": "train", "global_step": 739, "epoch": 0.029682291039081012, "loss": 0.0363, "grad_norm": 1.4499318599700928, "learning_rate": 7.763636363636364e-06, "num_tokens": 1631595.0, "completions/mean_length": 423.375, "completions/min_length": 398.0, "completions/max_length": 483.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 423.375, "completions/min_terminated_length": 398.0, "completions/max_terminated_length": 483.0, "rewards/meter/mean": 0.7534134387969971, "rewards/meter/std": 0.2928631901741028, "rewards/count_adherence/mean": 0.3888888955116272, "rewards/count_adherence/std": 0.059391383081674576, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5623973608016968, "rewards/repeat_penalty/std": 0.18699996173381805, "rewards/total_composite/mean": 0.15555939078330994, "rewards/total_composite/std": 0.07782954722642899, "reward": 0.15555939078330994, "reward_std": 0.07782954722642899, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013467294164001942, "sampling/sampling_logp_difference/max": 3.2660930156707764, "sampling/importance_sampling_ratio/min": 0.0381552055478096, "sampling/importance_sampling_ratio/mean": 0.9997313022613525, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0637149391695857, "clip_ratio/low_mean": 0.005539899080758914, "clip_ratio/low_min": 0.005539899080758914, "clip_ratio/high_mean": 0.004910050658509135, "clip_ratio/high_max": 0.004910050658509135, "clip_ratio/region_mean": 0.01044994973926805, "reward_total_mean": 0.15555939078330994, "reward_meter_mean": 0.7534134387969971, "reward_meter_std": 0.2928631901741028, "reward_count_adherence_mean": 0.3888888955116272, "reward_count_adherence_std": 0.059391383081674576, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5623973608016968, "reward_repeat_penalty_std": 0.18699996173381805, "reward_total_composite_mean": 0.15555939078330994, "reward_total_composite_std": 0.07782954722642899} {"timestamp_utc": "2026-04-11T22:49:19Z", "mode": "train", "global_step": 740, "epoch": 0.029722456520865966, "loss": -0.0085, "grad_norm": 3.1338510513305664, "learning_rate": 7.76060606060606e-06, "num_tokens": 1634096.0, "completions/mean_length": 133.625, "completions/min_length": 126.0, "completions/max_length": 142.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 133.625, "completions/min_terminated_length": 126.0, "completions/max_terminated_length": 142.0, "rewards/meter/mean": 0.11201652884483337, "rewards/meter/std": 0.19264455139636993, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6964285969734192, "rewards/repeat_penalty/std": 0.11921756714582443, "rewards/total_composite/mean": 0.0800618976354599, "rewards/total_composite/std": 0.13757191598415375, "reward": 0.0800618976354599, "reward_std": 0.13757191598415375, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02300534024834633, "sampling/sampling_logp_difference/max": 1.0942306518554688, "sampling/importance_sampling_ratio/min": 0.33479708433151245, "sampling/importance_sampling_ratio/mean": 1.005319595336914, "sampling/importance_sampling_ratio/max": 1.8425383567810059, "entropy": 0.15879025869071484, "clip_ratio/low_mean": 0.008456611772999167, "clip_ratio/low_min": 0.008456611772999167, "clip_ratio/high_mean": 0.009530608775094151, "clip_ratio/high_max": 0.009530608775094151, "clip_ratio/region_mean": 0.01798722054809332, "reward_total_mean": 0.0800618976354599, "reward_meter_mean": 0.11201652884483337, "reward_meter_std": 0.19264455139636993, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6964285969734192, "reward_repeat_penalty_std": 0.11921756714582443, "reward_total_composite_mean": 0.0800618976354599, "reward_total_composite_std": 0.13757191598415375} {"timestamp_utc": "2026-04-11T22:49:24Z", "mode": "train", "global_step": 741, "epoch": 0.02976262200265092, "loss": -0.0061, "grad_norm": 1.2720859050750732, "learning_rate": 7.757575757575758e-06, "num_tokens": 1636098.0, "completions/mean_length": 79.25, "completions/min_length": 78.0, "completions/max_length": 81.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.25, "completions/min_terminated_length": 78.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.9925205707550049, "rewards/meter/std": 0.0022588411811739206, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.794016420841217, "rewards/total_composite/std": 0.001807067426852882, "reward": 0.794016420841217, "reward_std": 0.0018070697551593184, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011411315761506557, "sampling/sampling_logp_difference/max": 0.8109602928161621, "sampling/importance_sampling_ratio/min": 0.4444310963153839, "sampling/importance_sampling_ratio/mean": 1.001147985458374, "sampling/importance_sampling_ratio/max": 1.5393544435501099, "entropy": 0.07466331776231527, "clip_ratio/low_mean": 0.006329114083200693, "clip_ratio/low_min": 0.006329114083200693, "clip_ratio/high_mean": 0.007794186705723405, "clip_ratio/high_max": 0.007794186705723405, "clip_ratio/region_mean": 0.014123300788924098, "reward_total_mean": 0.794016420841217, "reward_meter_mean": 0.9925205707550049, "reward_meter_std": 0.0022588411811739206, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.794016420841217, "reward_total_composite_std": 0.001807067426852882} {"timestamp_utc": "2026-04-11T22:49:29Z", "mode": "train", "global_step": 742, "epoch": 0.029802787484435874, "loss": 0.012, "grad_norm": 7.301284313201904, "learning_rate": 7.754545454545455e-06, "num_tokens": 1638424.0, "completions/mean_length": 120.75, "completions/min_length": 114.0, "completions/max_length": 125.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 120.75, "completions/min_terminated_length": 114.0, "completions/max_terminated_length": 125.0, "rewards/meter/mean": 0.7745586633682251, "rewards/meter/std": 0.25043216347694397, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6964285969734192, "rewards/repeat_penalty/std": 0.17806050181388855, "rewards/total_composite/mean": 0.5217785835266113, "rewards/total_composite/std": 0.1874629557132721, "reward": 0.5217785835266113, "reward_std": 0.1874629259109497, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04599983990192413, "sampling/sampling_logp_difference/max": 4.404999732971191, "sampling/importance_sampling_ratio/min": 0.012216109782457352, "sampling/importance_sampling_ratio/mean": 0.9962561726570129, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.12627906422130764, "clip_ratio/low_mean": 0.021951976465061307, "clip_ratio/low_min": 0.021951976465061307, "clip_ratio/high_mean": 0.018717249389737844, "clip_ratio/high_max": 0.018717249389737844, "clip_ratio/region_mean": 0.04066922585479915, "reward_total_mean": 0.5217785835266113, "reward_meter_mean": 0.7745586633682251, "reward_meter_std": 0.25043216347694397, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6964285969734192, "reward_repeat_penalty_std": 0.17806050181388855, "reward_total_composite_mean": 0.5217785835266113, "reward_total_composite_std": 0.1874629557132721} {"timestamp_utc": "2026-04-11T22:49:39Z", "mode": "train", "global_step": 743, "epoch": 0.02984295296622083, "loss": 0.1769, "grad_norm": 1.0547409057617188, "learning_rate": 7.751515151515153e-06, "num_tokens": 1642208.0, "completions/mean_length": 510.0, "completions/min_length": 507.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.5, "completions/mean_terminated_length": 508.0, "completions/min_terminated_length": 507.0, "completions/max_terminated_length": 510.0, "rewards/meter/mean": 0.45572322607040405, "rewards/meter/std": 0.1519794911146164, "rewards/count_adherence/mean": 0.6153846383094788, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.5480158925056458, "rewards/repeat_penalty/std": 0.010451768524944782, "rewards/total_composite/mean": 0.13946115970611572, "rewards/total_composite/std": 0.07740650326013565, "reward": 0.13946115970611572, "reward_std": 0.07740650326013565, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00907099712640047, "sampling/sampling_logp_difference/max": 1.135289192199707, "sampling/importance_sampling_ratio/min": 0.32132917642593384, "sampling/importance_sampling_ratio/mean": 1.0025880336761475, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03308352828025818, "clip_ratio/low_mean": 0.0017248676158487797, "clip_ratio/low_min": 0.0017248676158487797, "clip_ratio/high_mean": 0.0012283907853998244, "clip_ratio/high_max": 0.0012283907853998244, "clip_ratio/region_mean": 0.002953258401248604, "reward_total_mean": 0.13946115970611572, "reward_meter_mean": 0.45572322607040405, "reward_meter_std": 0.1519794911146164, "reward_count_adherence_mean": 0.6153846383094788, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.5480158925056458, "reward_repeat_penalty_std": 0.010451768524944782, "reward_total_composite_mean": 0.13946115970611572, "reward_total_composite_std": 0.07740650326013565} {"timestamp_utc": "2026-04-11T22:49:44Z", "mode": "train", "global_step": 744, "epoch": 0.029883118448005785, "loss": 0.0283, "grad_norm": 6.8046875, "learning_rate": 7.74848484848485e-06, "num_tokens": 1643996.0, "completions/mean_length": 64.5, "completions/min_length": 61.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.5, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.5396110415458679, "rewards/meter/std": 0.20649512112140656, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5396110415458679, "rewards/total_composite/std": 0.20649512112140656, "reward": 0.5396110415458679, "reward_std": 0.20649512112140656, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06342907249927521, "sampling/sampling_logp_difference/max": 1.2414631843566895, "sampling/importance_sampling_ratio/min": 0.28896111249923706, "sampling/importance_sampling_ratio/mean": 1.0097684860229492, "sampling/importance_sampling_ratio/max": 1.632293462753296, "entropy": 0.6070369817316532, "clip_ratio/low_mean": 0.02111415727995336, "clip_ratio/low_min": 0.02111415727995336, "clip_ratio/high_mean": 0.027501578675583005, "clip_ratio/high_max": 0.027501578675583005, "clip_ratio/region_mean": 0.048615735955536366, "reward_total_mean": 0.5396110415458679, "reward_meter_mean": 0.5396110415458679, "reward_meter_std": 0.20649512112140656, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5396110415458679, "reward_total_composite_std": 0.20649512112140656} {"timestamp_utc": "2026-04-11T22:49:49Z", "mode": "train", "global_step": 745, "epoch": 0.02992328392979074, "loss": 0.0099, "grad_norm": 9.705404281616211, "learning_rate": 7.745454545454545e-06, "num_tokens": 1645756.0, "completions/mean_length": 63.0, "completions/min_length": 62.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.0, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9103853106498718, "rewards/meter/std": 0.22165298461914062, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9103853106498718, "rewards/total_composite/std": 0.22165298461914062, "reward": 0.9103853106498718, "reward_std": 0.22165298461914062, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07189808040857315, "sampling/sampling_logp_difference/max": 2.013796806335449, "sampling/importance_sampling_ratio/min": 0.13348090648651123, "sampling/importance_sampling_ratio/mean": 1.0033738613128662, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.25149067025631666, "clip_ratio/low_mean": 0.008064515888690948, "clip_ratio/low_min": 0.008064515888690948, "clip_ratio/high_mean": 0.043645198456943035, "clip_ratio/high_max": 0.043645198456943035, "clip_ratio/region_mean": 0.051709714345633984, "reward_total_mean": 0.9103853106498718, "reward_meter_mean": 0.9103853106498718, "reward_meter_std": 0.22165298461914062, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9103853106498718, "reward_total_composite_std": 0.22165298461914062} {"timestamp_utc": "2026-04-11T22:49:56Z", "mode": "train", "global_step": 746, "epoch": 0.029963449411575693, "loss": 0.0227, "grad_norm": 2.2934165000915527, "learning_rate": 7.742424242424244e-06, "num_tokens": 1649182.0, "completions/mean_length": 238.25, "completions/min_length": 205.0, "completions/max_length": 260.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 238.25, "completions/min_terminated_length": 205.0, "completions/max_terminated_length": 260.0, "rewards/meter/mean": 0.8676279783248901, "rewards/meter/std": 0.3215867578983307, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.0707106739282608, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.36800700426101685, "rewards/repeat_penalty/std": 0.24577626585960388, "rewards/total_composite/mean": 0.2228499948978424, "rewards/total_composite/std": 0.18318407237529755, "reward": 0.2228499948978424, "reward_std": 0.18318407237529755, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018860070034861565, "sampling/sampling_logp_difference/max": 4.0910139083862305, "sampling/importance_sampling_ratio/min": 0.016722269356250763, "sampling/importance_sampling_ratio/mean": 1.0014268159866333, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06980151077732444, "clip_ratio/low_mean": 0.003769519622437656, "clip_ratio/low_min": 0.003769519622437656, "clip_ratio/high_mean": 0.006430621142499149, "clip_ratio/high_max": 0.006430621142499149, "clip_ratio/region_mean": 0.010200140764936805, "reward_total_mean": 0.2228499948978424, "reward_meter_mean": 0.8676279783248901, "reward_meter_std": 0.3215867578983307, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.0707106739282608, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.36800700426101685, "reward_repeat_penalty_std": 0.24577626585960388, "reward_total_composite_mean": 0.2228499948978424, "reward_total_composite_std": 0.18318407237529755} {"timestamp_utc": "2026-04-11T22:50:00Z", "mode": "train", "global_step": 747, "epoch": 0.030003614893360647, "loss": -0.0252, "grad_norm": 7.876811981201172, "learning_rate": 7.73939393939394e-06, "num_tokens": 1650683.0, "completions/mean_length": 40.625, "completions/min_length": 38.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.625, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.9663740396499634, "rewards/meter/std": 0.011263493448495865, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9663740396499634, "rewards/total_composite/std": 0.011263493448495865, "reward": 0.9663740396499634, "reward_std": 0.011263499036431313, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03788968175649643, "sampling/sampling_logp_difference/max": 0.8854336738586426, "sampling/importance_sampling_ratio/min": 0.41253525018692017, "sampling/importance_sampling_ratio/mean": 1.0029720067977905, "sampling/importance_sampling_ratio/max": 1.344473123550415, "entropy": 0.28702736645936966, "clip_ratio/low_mean": 0.012351190904155374, "clip_ratio/low_min": 0.012351190904155374, "clip_ratio/high_mean": 0.018227352295070887, "clip_ratio/high_max": 0.018227352295070887, "clip_ratio/region_mean": 0.03057854319922626, "reward_total_mean": 0.9663740396499634, "reward_meter_mean": 0.9663740396499634, "reward_meter_std": 0.011263493448495865, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9663740396499634, "reward_total_composite_std": 0.011263493448495865} {"timestamp_utc": "2026-04-11T22:50:05Z", "mode": "train", "global_step": 748, "epoch": 0.0300437803751456, "loss": 0.031, "grad_norm": 4.674757957458496, "learning_rate": 7.736363636363637e-06, "num_tokens": 1652564.0, "completions/mean_length": 82.125, "completions/min_length": 74.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.125, "completions/min_terminated_length": 74.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.8491703271865845, "rewards/meter/std": 0.15907908976078033, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8491703271865845, "rewards/total_composite/std": 0.15907908976078033, "reward": 0.8491703271865845, "reward_std": 0.15907907485961914, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0295439250767231, "sampling/sampling_logp_difference/max": 1.300436019897461, "sampling/importance_sampling_ratio/min": 0.2724129855632782, "sampling/importance_sampling_ratio/mean": 1.0036686658859253, "sampling/importance_sampling_ratio/max": 1.843922734260559, "entropy": 0.22462552785873413, "clip_ratio/low_mean": 0.002961171790957451, "clip_ratio/low_min": 0.002961171790957451, "clip_ratio/high_mean": 0.021474606008268893, "clip_ratio/high_max": 0.021474606008268893, "clip_ratio/region_mean": 0.024435777799226344, "reward_total_mean": 0.8491703271865845, "reward_meter_mean": 0.8491703271865845, "reward_meter_std": 0.15907908976078033, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8491703271865845, "reward_total_composite_std": 0.15907908976078033} {"timestamp_utc": "2026-04-11T22:50:10Z", "mode": "train", "global_step": 749, "epoch": 0.030083945856930555, "loss": 0.0372, "grad_norm": 6.799373626708984, "learning_rate": 7.733333333333334e-06, "num_tokens": 1654625.0, "completions/mean_length": 93.625, "completions/min_length": 87.0, "completions/max_length": 98.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.625, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 98.0, "rewards/meter/mean": 0.9564855694770813, "rewards/meter/std": 0.10235674679279327, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8500000238418579, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.8075714707374573, "rewards/total_composite/std": 0.0810769721865654, "reward": 0.8075714707374573, "reward_std": 0.0810769572854042, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03427453711628914, "sampling/sampling_logp_difference/max": 2.4858551025390625, "sampling/importance_sampling_ratio/min": 0.08325432986021042, "sampling/importance_sampling_ratio/mean": 0.9965777397155762, "sampling/importance_sampling_ratio/max": 1.903843641281128, "entropy": 0.12411046819761395, "clip_ratio/low_mean": 0.026110241888090968, "clip_ratio/low_min": 0.026110241888090968, "clip_ratio/high_mean": 0.008620689623057842, "clip_ratio/high_max": 0.008620689623057842, "clip_ratio/region_mean": 0.03473093151114881, "reward_total_mean": 0.8075714707374573, "reward_meter_mean": 0.9564855694770813, "reward_meter_std": 0.10235674679279327, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8500000238418579, "reward_repeat_penalty_std": 0.09258200973272324, "reward_total_composite_mean": 0.8075714707374573, "reward_total_composite_std": 0.0810769721865654} {"timestamp_utc": "2026-04-11T22:50:15Z", "mode": "train", "global_step": 750, "epoch": 0.03012411133871551, "loss": 0.031, "grad_norm": 6.688387870788574, "learning_rate": 7.730303030303032e-06, "num_tokens": 1656513.0, "completions/mean_length": 67.0, "completions/min_length": 65.0, "completions/max_length": 70.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.0, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 70.0, "rewards/meter/mean": 0.9406866431236267, "rewards/meter/std": 0.1574798822402954, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.8990964889526367, "rewards/total_composite/std": 0.18213681876659393, "reward": 0.8990964889526367, "reward_std": 0.18213684856891632, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.035663776099681854, "sampling/sampling_logp_difference/max": 0.9646925926208496, "sampling/importance_sampling_ratio/min": 0.42169255018234253, "sampling/importance_sampling_ratio/mean": 1.0121128559112549, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.21013184823095798, "clip_ratio/low_mean": 0.00892857147846371, "clip_ratio/low_min": 0.00892857147846371, "clip_ratio/high_mean": 0.022817256744019687, "clip_ratio/high_max": 0.022817256744019687, "clip_ratio/region_mean": 0.031745828222483397, "reward_total_mean": 0.8990964889526367, "reward_meter_mean": 0.9406866431236267, "reward_meter_std": 0.1574798822402954, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.8990964889526367, "reward_total_composite_std": 0.18213681876659393} {"timestamp_utc": "2026-04-11T22:51:48Z", "mode": "eval", "global_step": 750, "epoch": 0.03012411133871551, "eval_loss": NaN, "eval_runtime": 93.1633, "eval_samples_per_second": 1.116, "eval_steps_per_second": 0.14, "eval_num_tokens": 1656513.0, "eval_completions/mean_length": 284.52884615384613, "eval_completions/min_length": 61.46153846153846, "eval_completions/max_length": 496.7692307692308, "eval_completions/clipped_ratio": 0.23076923076923078, "eval_completions/mean_terminated_length": 215.78288092980017, "eval_completions/min_terminated_length": 61.46153846153846, "eval_completions/max_terminated_length": 393.46153846153845, "eval_rewards/meter/mean": 0.6660121427132533, "eval_rewards/meter/std": 0.37975076070198643, "eval_rewards/count_adherence/mean": 0.7424245018225449, "eval_rewards/count_adherence/std": 0.2582179422561939, "eval_rewards/arabic_clean/mean": 0.9711538461538461, "eval_rewards/arabic_clean/std": 0.08158924258672275, "eval_rewards/repeat_penalty/mean": 0.638540084545429, "eval_rewards/repeat_penalty/std": 0.270676647241299, "eval_rewards/total_composite/mean": 0.33899185634576356, "eval_rewards/total_composite/std": 0.3308297275350644, "eval_reward": 0.33899185634576356, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.012646582407447008, "eval_sampling/sampling_logp_difference/max": 0.8113565261547382, "eval_sampling/importance_sampling_ratio/min": 0.4615517258644104, "eval_sampling/importance_sampling_ratio/mean": 1.0036690326837392, "eval_sampling/importance_sampling_ratio/max": 1.3942875678722675, "eval_entropy": 0.17351046003974402, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.33899185634576356, "eval_reward_meter_mean": 0.6660121427132533, "eval_reward_meter_std": 0.37975076070198643, "eval_reward_count_adherence_mean": 0.7424245018225449, "eval_reward_count_adherence_std": 0.2582179422561939, "eval_reward_arabic_clean_mean": 0.9711538461538461, "eval_reward_arabic_clean_std": 0.08158924258672275, "eval_reward_repeat_penalty_mean": 0.638540084545429, "eval_reward_repeat_penalty_std": 0.270676647241299, "eval_reward_total_composite_mean": 0.33899185634576356, "eval_reward_total_composite_std": 0.3308297275350644} {"timestamp_utc": "2026-04-11T22:51:55Z", "mode": "train", "global_step": 751, "epoch": 0.030164276820500463, "loss": 0.0024, "grad_norm": 5.31432580947876, "learning_rate": 7.727272727272727e-06, "num_tokens": 1658395.0, "completions/mean_length": 59.25, "completions/min_length": 59.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.25, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9893741607666016, "rewards/meter/std": 0.007569636683911085, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9893741607666016, "rewards/total_composite/std": 0.007569636683911085, "reward": 0.9893741607666016, "reward_std": 0.007569642271846533, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013296845369040966, "sampling/sampling_logp_difference/max": 0.9290802478790283, "sampling/importance_sampling_ratio/min": 0.3949167728424072, "sampling/importance_sampling_ratio/mean": 1.0002750158309937, "sampling/importance_sampling_ratio/max": 1.868045449256897, "entropy": 0.052004152443259954, "clip_ratio/low_mean": 0.006320621585473418, "clip_ratio/low_min": 0.006320621585473418, "clip_ratio/high_mean": 0.008403955027461052, "clip_ratio/high_max": 0.008403955027461052, "clip_ratio/region_mean": 0.01472457661293447, "reward_total_mean": 0.9893741607666016, "reward_meter_mean": 0.9893741607666016, "reward_meter_std": 0.007569636683911085, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9893741607666016, "reward_total_composite_std": 0.007569636683911085} {"timestamp_utc": "2026-04-11T22:52:00Z", "mode": "train", "global_step": 752, "epoch": 0.030204442302285417, "loss": 0.0051, "grad_norm": 4.07772970199585, "learning_rate": 7.724242424242424e-06, "num_tokens": 1660117.0, "completions/mean_length": 59.25, "completions/min_length": 59.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.25, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9949396848678589, "rewards/meter/std": 0.00028474157443270087, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9534660577774048, "rewards/total_composite/std": 0.11713220924139023, "reward": 0.9534660577774048, "reward_std": 0.11713218688964844, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009837880730628967, "sampling/sampling_logp_difference/max": 1.2378616333007812, "sampling/importance_sampling_ratio/min": 0.2900037169456482, "sampling/importance_sampling_ratio/mean": 1.0019704103469849, "sampling/importance_sampling_ratio/max": 1.3358639478683472, "entropy": 0.03427604655735195, "clip_ratio/low_mean": 0.0020833334419876337, "clip_ratio/low_min": 0.0020833334419876337, "clip_ratio/high_mean": 0.006320621585473418, "clip_ratio/high_max": 0.006320621585473418, "clip_ratio/region_mean": 0.008403955027461052, "reward_total_mean": 0.9534660577774048, "reward_meter_mean": 0.9949396848678589, "reward_meter_std": 0.00028474157443270087, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.9534660577774048, "reward_total_composite_std": 0.11713220924139023} {"timestamp_utc": "2026-04-11T22:52:04Z", "mode": "train", "global_step": 753, "epoch": 0.03024460778407037, "loss": -0.0196, "grad_norm": 14.107855796813965, "learning_rate": 7.721212121212122e-06, "num_tokens": 1661428.0, "completions/mean_length": 35.875, "completions/min_length": 34.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.875, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.9349486827850342, "rewards/meter/std": 0.1060987114906311, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9349486827850342, "rewards/total_composite/std": 0.1060987114906311, "reward": 0.9349486827850342, "reward_std": 0.1060987114906311, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07551710307598114, "sampling/sampling_logp_difference/max": 1.5768651962280273, "sampling/importance_sampling_ratio/min": 0.20662181079387665, "sampling/importance_sampling_ratio/mean": 1.0009469985961914, "sampling/importance_sampling_ratio/max": 1.6115726232528687, "entropy": 0.6193088293075562, "clip_ratio/low_mean": 0.017439668532460928, "clip_ratio/low_min": 0.017439668532460928, "clip_ratio/high_mean": 0.04496192745864391, "clip_ratio/high_max": 0.04496192745864391, "clip_ratio/region_mean": 0.06240159599110484, "reward_total_mean": 0.9349486827850342, "reward_meter_mean": 0.9349486827850342, "reward_meter_std": 0.1060987114906311, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9349486827850342, "reward_total_composite_std": 0.1060987114906311} {"timestamp_utc": "2026-04-11T22:52:09Z", "mode": "train", "global_step": 754, "epoch": 0.030284773265855325, "loss": 0.0091, "grad_norm": 2.418649673461914, "learning_rate": 7.718181818181819e-06, "num_tokens": 1663080.0, "completions/mean_length": 41.5, "completions/min_length": 41.0, "completions/max_length": 42.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.5, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 42.0, "rewards/meter/mean": 0.9943797588348389, "rewards/meter/std": 0.0011224248446524143, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9943797588348389, "rewards/total_composite/std": 0.0011224248446524143, "reward": 0.9943797588348389, "reward_std": 0.0011224271729588509, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.030370449647307396, "sampling/sampling_logp_difference/max": 1.4918346405029297, "sampling/importance_sampling_ratio/min": 0.2249595671892166, "sampling/importance_sampling_ratio/mean": 0.9925946593284607, "sampling/importance_sampling_ratio/max": 1.2741270065307617, "entropy": 0.11915541160851717, "clip_ratio/low_mean": 0.008928571594879031, "clip_ratio/low_min": 0.008928571594879031, "clip_ratio/high_mean": 0.018074912950396538, "clip_ratio/high_max": 0.018074912950396538, "clip_ratio/region_mean": 0.02700348454527557, "reward_total_mean": 0.9943797588348389, "reward_meter_mean": 0.9943797588348389, "reward_meter_std": 0.0011224248446524143, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9943797588348389, "reward_total_composite_std": 0.0011224248446524143} {"timestamp_utc": "2026-04-11T22:52:13Z", "mode": "train", "global_step": 755, "epoch": 0.03032493874764028, "loss": -0.0018, "grad_norm": 5.144253730773926, "learning_rate": 7.715151515151516e-06, "num_tokens": 1664616.0, "completions/mean_length": 41.0, "completions/min_length": 41.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.0, "completions/min_terminated_length": 41.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.9944278001785278, "rewards/meter/std": 0.0012945194030180573, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9944278001785278, "rewards/total_composite/std": 0.0012945194030180573, "reward": 0.9944278001785278, "reward_std": 0.0012945224298164248, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014835118316113949, "sampling/sampling_logp_difference/max": 0.617079496383667, "sampling/importance_sampling_ratio/min": 0.5395178198814392, "sampling/importance_sampling_ratio/mean": 1.0021907091140747, "sampling/importance_sampling_ratio/max": 1.2102103233337402, "entropy": 0.10165042616426945, "clip_ratio/low_mean": 0.0030487803742289543, "clip_ratio/low_min": 0.0030487803742289543, "clip_ratio/high_mean": 0.012195121496915817, "clip_ratio/high_max": 0.012195121496915817, "clip_ratio/region_mean": 0.015243901871144772, "reward_total_mean": 0.9944278001785278, "reward_meter_mean": 0.9944278001785278, "reward_meter_std": 0.0012945194030180573, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9944278001785278, "reward_total_composite_std": 0.0012945194030180573} {"timestamp_utc": "2026-04-11T22:52:19Z", "mode": "train", "global_step": 756, "epoch": 0.030365104229425233, "loss": -0.0099, "grad_norm": 2.4242031574249268, "learning_rate": 7.712121212121213e-06, "num_tokens": 1666878.0, "completions/mean_length": 113.75, "completions/min_length": 111.0, "completions/max_length": 120.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 113.75, "completions/min_terminated_length": 111.0, "completions/max_terminated_length": 120.0, "rewards/meter/mean": 0.9662827253341675, "rewards/meter/std": 0.009305375628173351, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6500000357627869, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.6285786032676697, "rewards/total_composite/std": 0.0941418930888176, "reward": 0.6285786032676697, "reward_std": 0.09414192289113998, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011911381967365742, "sampling/sampling_logp_difference/max": 1.6042141914367676, "sampling/importance_sampling_ratio/min": 0.20104746520519257, "sampling/importance_sampling_ratio/mean": 1.0027672052383423, "sampling/importance_sampling_ratio/max": 1.911029577255249, "entropy": 0.0643842932768166, "clip_ratio/low_mean": 0.005512091098353267, "clip_ratio/low_min": 0.005512091098353267, "clip_ratio/high_mean": 0.005401917500421405, "clip_ratio/high_max": 0.005401917500421405, "clip_ratio/region_mean": 0.010914008598774672, "reward_total_mean": 0.6285786032676697, "reward_meter_mean": 0.9662827253341675, "reward_meter_std": 0.009305375628173351, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6500000357627869, "reward_repeat_penalty_std": 0.09258200973272324, "reward_total_composite_mean": 0.6285786032676697, "reward_total_composite_std": 0.0941418930888176} {"timestamp_utc": "2026-04-11T22:52:26Z", "mode": "train", "global_step": 757, "epoch": 0.030405269711210187, "loss": 0.0259, "grad_norm": 1.2799259424209595, "learning_rate": 7.709090909090909e-06, "num_tokens": 1670368.0, "completions/mean_length": 243.25, "completions/min_length": 217.0, "completions/max_length": 267.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 243.25, "completions/min_terminated_length": 217.0, "completions/max_terminated_length": 267.0, "rewards/meter/mean": 0.9969884157180786, "rewards/meter/std": 0.000910833477973938, "rewards/count_adherence/mean": 0.6500000357627869, "rewards/count_adherence/std": 0.09258200973272324, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.4187062978744507, "rewards/repeat_penalty/std": 0.23771634697914124, "rewards/total_composite/mean": 0.2822417914867401, "rewards/total_composite/std": 0.18207497894763947, "reward": 0.2822417914867401, "reward_std": 0.18207496404647827, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00974962953478098, "sampling/sampling_logp_difference/max": 1.2791423797607422, "sampling/importance_sampling_ratio/min": 0.27827587723731995, "sampling/importance_sampling_ratio/mean": 1.0003676414489746, "sampling/importance_sampling_ratio/max": 1.5265671014785767, "entropy": 0.06725997012108564, "clip_ratio/low_mean": 0.0010245901066809893, "clip_ratio/low_min": 0.0010245901066809893, "clip_ratio/high_mean": 0.007282168400706723, "clip_ratio/high_max": 0.007282168400706723, "clip_ratio/region_mean": 0.008306758507387713, "reward_total_mean": 0.2822417914867401, "reward_meter_mean": 0.9969884157180786, "reward_meter_std": 0.000910833477973938, "reward_count_adherence_mean": 0.6500000357627869, "reward_count_adherence_std": 0.09258200973272324, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.4187062978744507, "reward_repeat_penalty_std": 0.23771634697914124, "reward_total_composite_mean": 0.2822417914867401, "reward_total_composite_std": 0.18207497894763947} {"timestamp_utc": "2026-04-11T22:52:37Z", "mode": "train", "global_step": 758, "epoch": 0.03044543519299514, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.706060606060606e-06, "num_tokens": 1672112.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.8747541308403015, "rewards/meter/std": 0.1777074635028839, "rewards/count_adherence/mean": 0.15625, "rewards/count_adherence/std": 0.11080066114664078, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.42480412125587463, "rewards/repeat_penalty/std": 0.2929826080799103, "rewards/total_composite/mean": 0.059182293713092804, "rewards/total_composite/std": 0.06593845039606094, "reward": 0.059182293713092804, "reward_std": 0.06593845039606094, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.059182293713092804, "reward_meter_mean": 0.8747541308403015, "reward_meter_std": 0.1777074635028839, "reward_count_adherence_mean": 0.15625, "reward_count_adherence_std": 0.11080066114664078, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.42480412125587463, "reward_repeat_penalty_std": 0.2929826080799103, "reward_total_composite_mean": 0.059182293713092804, "reward_total_composite_std": 0.06593845039606094} {"timestamp_utc": "2026-04-11T22:52:42Z", "mode": "train", "global_step": 759, "epoch": 0.030485600674780094, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.703030303030304e-06, "num_tokens": 1674144.0, "completions/mean_length": 87.0, "completions/min_length": 87.0, "completions/max_length": 87.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.0, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.9949710369110107, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6000000238418579, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5969825983047485, "rewards/total_composite/std": 0.0, "reward": 0.5969825983047485, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.001654994674026966, "sampling/sampling_logp_difference/max": 0.1641908884048462, "sampling/importance_sampling_ratio/min": 0.848580002784729, "sampling/importance_sampling_ratio/mean": 1.0006695985794067, "sampling/importance_sampling_ratio/max": 1.0613410472869873, "entropy": 0.014668526826426387, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.5969825983047485, "reward_meter_mean": 0.9949710369110107, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6000000238418579, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5969825983047485, "reward_total_composite_std": 0.0} {"timestamp_utc": "2026-04-11T22:52:46Z", "mode": "train", "global_step": 760, "epoch": 0.03052576615656505, "loss": -0.0037, "grad_norm": 4.493424415588379, "learning_rate": 7.7e-06, "num_tokens": 1675945.0, "completions/mean_length": 54.125, "completions/min_length": 54.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.125, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.9517734050750732, "rewards/meter/std": 0.0011382178636267781, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7083333730697632, "rewards/repeat_penalty/std": 0.11785111576318741, "rewards/total_composite/mean": 0.6740554571151733, "rewards/total_composite/std": 0.11107677966356277, "reward": 0.6740554571151733, "reward_std": 0.11107677221298218, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005725136026740074, "sampling/sampling_logp_difference/max": 0.4777810573577881, "sampling/importance_sampling_ratio/min": 0.7561957836151123, "sampling/importance_sampling_ratio/mean": 1.0033941268920898, "sampling/importance_sampling_ratio/max": 1.612492322921753, "entropy": 0.042751661501824856, "clip_ratio/low_mean": 0.004629629664123058, "clip_ratio/low_min": 0.004629629664123058, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.004629629664123058, "reward_total_mean": 0.6740554571151733, "reward_meter_mean": 0.9517734050750732, "reward_meter_std": 0.0011382178636267781, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7083333730697632, "reward_repeat_penalty_std": 0.11785111576318741, "reward_total_composite_mean": 0.6740554571151733, "reward_total_composite_std": 0.11107677966356277} {"timestamp_utc": "2026-04-11T22:52:51Z", "mode": "train", "global_step": 761, "epoch": 0.030565931638350002, "loss": -0.0023, "grad_norm": 0.8903390765190125, "learning_rate": 7.696969696969696e-06, "num_tokens": 1677747.0, "completions/mean_length": 58.25, "completions/min_length": 58.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.25, "completions/min_terminated_length": 58.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.9949068427085876, "rewards/meter/std": 0.0007968654972501099, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7083333730697632, "rewards/repeat_penalty/std": 0.11785111576318741, "rewards/total_composite/mean": 0.7047345042228699, "rewards/total_composite/std": 0.1173337996006012, "reward": 0.7047345042228699, "reward_std": 0.1173337996006012, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011640225537121296, "sampling/sampling_logp_difference/max": 1.0540056228637695, "sampling/importance_sampling_ratio/min": 0.3485388159751892, "sampling/importance_sampling_ratio/mean": 1.0014004707336426, "sampling/importance_sampling_ratio/max": 1.6007080078125, "entropy": 0.05586709058843553, "clip_ratio/low_mean": 0.006428988883271813, "clip_ratio/low_min": 0.006428988883271813, "clip_ratio/high_mean": 0.0021186440717428923, "clip_ratio/high_max": 0.0021186440717428923, "clip_ratio/region_mean": 0.008547632955014706, "reward_total_mean": 0.7047345042228699, "reward_meter_mean": 0.9949068427085876, "reward_meter_std": 0.0007968654972501099, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7083333730697632, "reward_repeat_penalty_std": 0.11785111576318741, "reward_total_composite_mean": 0.7047345042228699, "reward_total_composite_std": 0.1173337996006012} {"timestamp_utc": "2026-04-11T22:52:56Z", "mode": "train", "global_step": 762, "epoch": 0.030606097120134956, "loss": 0.024, "grad_norm": 5.6463236808776855, "learning_rate": 7.693939393939395e-06, "num_tokens": 1679613.0, "completions/mean_length": 78.25, "completions/min_length": 75.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 78.25, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.8964411020278931, "rewards/meter/std": 0.12688924372196198, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6750000715255737, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.6016985774040222, "rewards/total_composite/std": 0.1109333410859108, "reward": 0.6016985774040222, "reward_std": 0.11093335598707199, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02804330736398697, "sampling/sampling_logp_difference/max": 0.9042620658874512, "sampling/importance_sampling_ratio/min": 0.4048405587673187, "sampling/importance_sampling_ratio/mean": 1.0003811120986938, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11404913989827037, "clip_ratio/low_mean": 0.012610982405021787, "clip_ratio/low_min": 0.012610982405021787, "clip_ratio/high_mean": 0.006666666595265269, "clip_ratio/high_max": 0.006666666595265269, "clip_ratio/region_mean": 0.019277649000287056, "reward_total_mean": 0.6016985774040222, "reward_meter_mean": 0.8964411020278931, "reward_meter_std": 0.12688924372196198, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6750000715255737, "reward_repeat_penalty_std": 0.1035098284482956, "reward_total_composite_mean": 0.6016985774040222, "reward_total_composite_std": 0.1109333410859108} {"timestamp_utc": "2026-04-11T22:53:06Z", "mode": "train", "global_step": 763, "epoch": 0.03064626260191991, "loss": 0.0481, "grad_norm": 1.2447540760040283, "learning_rate": 7.690909090909091e-06, "num_tokens": 1684742.0, "completions/mean_length": 438.125, "completions/min_length": 391.0, "completions/max_length": 498.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 438.125, "completions/min_terminated_length": 391.0, "completions/max_terminated_length": 498.0, "rewards/meter/mean": 0.9878873825073242, "rewards/meter/std": 0.019713442772626877, "rewards/count_adherence/mean": 0.2142857164144516, "rewards/count_adherence/std": 0.07636035978794098, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5576170682907104, "rewards/repeat_penalty/std": 0.0946657732129097, "rewards/total_composite/mean": 0.12146620452404022, "rewards/total_composite/std": 0.05504889413714409, "reward": 0.12146620452404022, "reward_std": 0.05504889413714409, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010950884781777859, "sampling/sampling_logp_difference/max": 1.4895976781845093, "sampling/importance_sampling_ratio/min": 0.22546334564685822, "sampling/importance_sampling_ratio/mean": 1.0031057596206665, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07848789915442467, "clip_ratio/low_mean": 0.005670643062330782, "clip_ratio/low_min": 0.005670643062330782, "clip_ratio/high_mean": 0.005744358117226511, "clip_ratio/high_max": 0.005744358117226511, "clip_ratio/region_mean": 0.011415001179557294, "reward_total_mean": 0.12146620452404022, "reward_meter_mean": 0.9878873825073242, "reward_meter_std": 0.019713442772626877, "reward_count_adherence_mean": 0.2142857164144516, "reward_count_adherence_std": 0.07636035978794098, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5576170682907104, "reward_repeat_penalty_std": 0.0946657732129097, "reward_total_composite_mean": 0.12146620452404022, "reward_total_composite_std": 0.05504889413714409} {"timestamp_utc": "2026-04-11T22:53:11Z", "mode": "train", "global_step": 764, "epoch": 0.030686428083704864, "loss": 0.0189, "grad_norm": 5.640087127685547, "learning_rate": 7.687878787878788e-06, "num_tokens": 1686539.0, "completions/mean_length": 63.625, "completions/min_length": 60.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.625, "completions/min_terminated_length": 60.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.874701976776123, "rewards/meter/std": 0.18508003652095795, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.7850605845451355, "rewards/total_composite/std": 0.2699912190437317, "reward": 0.7850605845451355, "reward_std": 0.2699912190437317, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.032904475927352905, "sampling/sampling_logp_difference/max": 1.8050861358642578, "sampling/importance_sampling_ratio/min": 0.16446030139923096, "sampling/importance_sampling_ratio/mean": 0.9924401044845581, "sampling/importance_sampling_ratio/max": 1.438651442527771, "entropy": 0.1180117940530181, "clip_ratio/low_mean": 0.005871212342754006, "clip_ratio/low_min": 0.005871212342754006, "clip_ratio/high_mean": 0.02714023506268859, "clip_ratio/high_max": 0.02714023506268859, "clip_ratio/region_mean": 0.033011447405442595, "reward_total_mean": 0.7850605845451355, "reward_meter_mean": 0.874701976776123, "reward_meter_std": 0.18508003652095795, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.17251639068126678, "reward_total_composite_mean": 0.7850605845451355, "reward_total_composite_std": 0.2699912190437317} {"timestamp_utc": "2026-04-11T22:53:16Z", "mode": "train", "global_step": 765, "epoch": 0.030726593565489818, "loss": 0.0008, "grad_norm": 4.968373775482178, "learning_rate": 7.684848484848485e-06, "num_tokens": 1688548.0, "completions/mean_length": 87.125, "completions/min_length": 87.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.125, "completions/min_terminated_length": 87.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.9950021505355835, "rewards/meter/std": 8.806584810372442e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6500000357627869, "rewards/repeat_penalty/std": 0.1414213478565216, "rewards/total_composite/mean": 0.6467622518539429, "rewards/total_composite/std": 0.14079822599887848, "reward": 0.6467622518539429, "reward_std": 0.14079821109771729, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004592791199684143, "sampling/sampling_logp_difference/max": 1.7399308681488037, "sampling/importance_sampling_ratio/min": 0.1755325347185135, "sampling/importance_sampling_ratio/mean": 0.9999979138374329, "sampling/importance_sampling_ratio/max": 1.767101764678955, "entropy": 0.008648848335724324, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0028409091755747795, "clip_ratio/high_max": 0.0028409091755747795, "clip_ratio/region_mean": 0.0028409091755747795, "reward_total_mean": 0.6467622518539429, "reward_meter_mean": 0.9950021505355835, "reward_meter_std": 8.806584810372442e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6500000357627869, "reward_repeat_penalty_std": 0.1414213478565216, "reward_total_composite_mean": 0.6467622518539429, "reward_total_composite_std": 0.14079822599887848} {"timestamp_utc": "2026-04-11T22:53:21Z", "mode": "train", "global_step": 766, "epoch": 0.030766759047274772, "loss": 0.0014, "grad_norm": 10.79392147064209, "learning_rate": 7.681818181818183e-06, "num_tokens": 1690236.0, "completions/mean_length": 52.0, "completions/min_length": 51.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.0, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.8247180581092834, "rewards/meter/std": 0.3005145490169525, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8247180581092834, "rewards/total_composite/std": 0.3005145490169525, "reward": 0.8247180581092834, "reward_std": 0.3005145490169525, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04232970252633095, "sampling/sampling_logp_difference/max": 1.6089832782745361, "sampling/importance_sampling_ratio/min": 0.2000909447669983, "sampling/importance_sampling_ratio/mean": 0.997559666633606, "sampling/importance_sampling_ratio/max": 1.6643987894058228, "entropy": 0.19545143470168114, "clip_ratio/low_mean": 0.004807692486792803, "clip_ratio/low_min": 0.004807692486792803, "clip_ratio/high_mean": 0.03126057400368154, "clip_ratio/high_max": 0.03126057400368154, "clip_ratio/region_mean": 0.03606826649047434, "reward_total_mean": 0.8247180581092834, "reward_meter_mean": 0.8247180581092834, "reward_meter_std": 0.3005145490169525, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8247180581092834, "reward_total_composite_std": 0.3005145490169525} {"timestamp_utc": "2026-04-11T22:53:26Z", "mode": "train", "global_step": 767, "epoch": 0.030806924529059726, "loss": 0.0103, "grad_norm": 12.172296524047852, "learning_rate": 7.678787878787878e-06, "num_tokens": 1692065.0, "completions/mean_length": 72.625, "completions/min_length": 67.0, "completions/max_length": 77.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.625, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 77.0, "rewards/meter/mean": 0.3632194995880127, "rewards/meter/std": 0.32341283559799194, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.36088046431541443, "rewards/total_composite/std": 0.3263150453567505, "reward": 0.36088046431541443, "reward_std": 0.3263150453567505, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07868395745754242, "sampling/sampling_logp_difference/max": 5.154880046844482, "sampling/importance_sampling_ratio/min": 0.005771172232925892, "sampling/importance_sampling_ratio/mean": 0.9972518086433411, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.46072521805763245, "clip_ratio/low_mean": 0.03220835281535983, "clip_ratio/low_min": 0.03220835281535983, "clip_ratio/high_mean": 0.026226354064419866, "clip_ratio/high_max": 0.026226354064419866, "clip_ratio/region_mean": 0.058434706879779696, "reward_total_mean": 0.36088046431541443, "reward_meter_mean": 0.3632194995880127, "reward_meter_std": 0.32341283559799194, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.36088046431541443, "reward_total_composite_std": 0.3263150453567505} {"timestamp_utc": "2026-04-11T22:53:31Z", "mode": "train", "global_step": 768, "epoch": 0.03084709001084468, "loss": -0.0095, "grad_norm": 2.0709922313690186, "learning_rate": 7.675757575757577e-06, "num_tokens": 1694344.0, "completions/mean_length": 107.875, "completions/min_length": 104.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.875, "completions/min_terminated_length": 104.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.7045435905456543, "rewards/meter/std": 0.13543730974197388, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5636348724365234, "rewards/total_composite/std": 0.10834985971450806, "reward": 0.5636348724365234, "reward_std": 0.10834983736276627, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010335095226764679, "sampling/sampling_logp_difference/max": 0.9596023559570312, "sampling/importance_sampling_ratio/min": 0.38304516673088074, "sampling/importance_sampling_ratio/mean": 1.0031026601791382, "sampling/importance_sampling_ratio/max": 1.4763460159301758, "entropy": 0.09003173373639584, "clip_ratio/low_mean": 0.0059091257862746716, "clip_ratio/low_min": 0.0059091257862746716, "clip_ratio/high_mean": 0.0022935778833925724, "clip_ratio/high_max": 0.0022935778833925724, "clip_ratio/region_mean": 0.008202703669667244, "reward_total_mean": 0.5636348724365234, "reward_meter_mean": 0.7045435905456543, "reward_meter_std": 0.13543730974197388, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5636348724365234, "reward_total_composite_std": 0.10834985971450806} {"timestamp_utc": "2026-04-11T22:53:37Z", "mode": "train", "global_step": 769, "epoch": 0.030887255492629634, "loss": 0.0714, "grad_norm": 7.707916736602783, "learning_rate": 7.672727272727273e-06, "num_tokens": 1696973.0, "completions/mean_length": 143.625, "completions/min_length": 126.0, "completions/max_length": 172.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 143.625, "completions/min_terminated_length": 126.0, "completions/max_terminated_length": 172.0, "rewards/meter/mean": 0.7441333532333374, "rewards/meter/std": 0.4554755687713623, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.10690449178218842, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6174242496490479, "rewards/repeat_penalty/std": 0.1036364957690239, "rewards/total_composite/mean": 0.4105571210384369, "rewards/total_composite/std": 0.2803230583667755, "reward": 0.4105571210384369, "reward_std": 0.2803230583667755, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03509281575679779, "sampling/sampling_logp_difference/max": 4.862173557281494, "sampling/importance_sampling_ratio/min": 0.0077336556278169155, "sampling/importance_sampling_ratio/mean": 0.9966462850570679, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0793102509342134, "clip_ratio/low_mean": 0.009834110038354993, "clip_ratio/low_min": 0.009834110038354993, "clip_ratio/high_mean": 0.0102224723668769, "clip_ratio/high_max": 0.0102224723668769, "clip_ratio/region_mean": 0.020056582405231893, "reward_total_mean": 0.4105571210384369, "reward_meter_mean": 0.7441333532333374, "reward_meter_std": 0.4554755687713623, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.10690449178218842, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6174242496490479, "reward_repeat_penalty_std": 0.1036364957690239, "reward_total_composite_mean": 0.4105571210384369, "reward_total_composite_std": 0.2803230583667755} {"timestamp_utc": "2026-04-11T22:53:47Z", "mode": "train", "global_step": 770, "epoch": 0.030927420974414588, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.66969696969697e-06, "num_tokens": 1698965.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.8939934968948364, "rewards/meter/std": 0.13721102476119995, "rewards/count_adherence/mean": 0.7828947305679321, "rewards/count_adherence/std": 0.03373000770807266, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5402884483337402, "rewards/repeat_penalty/std": 0.2157072126865387, "rewards/total_composite/mean": 0.3744324743747711, "rewards/total_composite/std": 0.17441345751285553, "reward": 0.3744324743747711, "reward_std": 0.17441345751285553, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.3744324743747711, "reward_meter_mean": 0.8939934968948364, "reward_meter_std": 0.13721102476119995, "reward_count_adherence_mean": 0.7828947305679321, "reward_count_adherence_std": 0.03373000770807266, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5402884483337402, "reward_repeat_penalty_std": 0.2157072126865387, "reward_total_composite_mean": 0.3744324743747711, "reward_total_composite_std": 0.17441345751285553} {"timestamp_utc": "2026-04-11T22:53:54Z", "mode": "train", "global_step": 771, "epoch": 0.030967586456199542, "loss": 0.0526, "grad_norm": 4.9908342361450195, "learning_rate": 7.666666666666667e-06, "num_tokens": 1702409.0, "completions/mean_length": 230.5, "completions/min_length": 180.0, "completions/max_length": 263.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 230.5, "completions/min_terminated_length": 180.0, "completions/max_terminated_length": 263.0, "rewards/meter/mean": 0.7852883338928223, "rewards/meter/std": 0.3382474482059479, "rewards/count_adherence/mean": 0.8250000476837158, "rewards/count_adherence/std": 0.0707106739282608, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5688130855560303, "rewards/repeat_penalty/std": 0.20688475668430328, "rewards/total_composite/mean": 0.3491661250591278, "rewards/total_composite/std": 0.22600221633911133, "reward": 0.3491661250591278, "reward_std": 0.22600221633911133, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018919790163636208, "sampling/sampling_logp_difference/max": 1.5345475673675537, "sampling/importance_sampling_ratio/min": 0.21555320918560028, "sampling/importance_sampling_ratio/mean": 1.0030508041381836, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.13555688876658678, "clip_ratio/low_mean": 0.004294316866435111, "clip_ratio/low_min": 0.004294316866435111, "clip_ratio/high_mean": 0.010813763190526515, "clip_ratio/high_max": 0.010813763190526515, "clip_ratio/region_mean": 0.015108080056961626, "reward_total_mean": 0.3491661250591278, "reward_meter_mean": 0.7852883338928223, "reward_meter_std": 0.3382474482059479, "reward_count_adherence_mean": 0.8250000476837158, "reward_count_adherence_std": 0.0707106739282608, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5688130855560303, "reward_repeat_penalty_std": 0.20688475668430328, "reward_total_composite_mean": 0.3491661250591278, "reward_total_composite_std": 0.22600221633911133} {"timestamp_utc": "2026-04-11T22:54:00Z", "mode": "train", "global_step": 772, "epoch": 0.031007751937984496, "loss": -0.1232, "grad_norm": 5.677700042724609, "learning_rate": 7.663636363636364e-06, "num_tokens": 1704516.0, "completions/mean_length": 99.375, "completions/min_length": 90.0, "completions/max_length": 123.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 99.375, "completions/min_terminated_length": 90.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.9852504730224609, "rewards/meter/std": 0.008375532925128937, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.1157275140285492, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7785714864730835, "rewards/repeat_penalty/std": 0.039677999913692474, "rewards/total_composite/mean": 0.6195021271705627, "rewards/total_composite/std": 0.05528084933757782, "reward": 0.6195021271705627, "reward_std": 0.05528085306286812, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020249057561159134, "sampling/sampling_logp_difference/max": 1.5644612312316895, "sampling/importance_sampling_ratio/min": 0.2092006951570511, "sampling/importance_sampling_ratio/mean": 1.0006357431411743, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.08717024885118008, "clip_ratio/low_mean": 0.015005706925876439, "clip_ratio/low_min": 0.015005706925876439, "clip_ratio/high_mean": 0.006097560748457909, "clip_ratio/high_max": 0.006097560748457909, "clip_ratio/region_mean": 0.021103267674334347, "reward_total_mean": 0.6195021271705627, "reward_meter_mean": 0.9852504730224609, "reward_meter_std": 0.008375532925128937, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.1157275140285492, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7785714864730835, "reward_repeat_penalty_std": 0.039677999913692474, "reward_total_composite_mean": 0.6195021271705627, "reward_total_composite_std": 0.05528084933757782} {"timestamp_utc": "2026-04-11T22:54:09Z", "mode": "train", "global_step": 773, "epoch": 0.03104791741976945, "loss": 0.0245, "grad_norm": 5.519010066986084, "learning_rate": 7.660606060606062e-06, "num_tokens": 1709201.0, "completions/mean_length": 409.625, "completions/min_length": 346.0, "completions/max_length": 460.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 409.625, "completions/min_terminated_length": 346.0, "completions/max_terminated_length": 460.0, "rewards/meter/mean": 0.9907213449478149, "rewards/meter/std": 0.006963523104786873, "rewards/count_adherence/mean": 0.3571428656578064, "rewards/count_adherence/std": 0.07636035233736038, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.45148104429244995, "rewards/repeat_penalty/std": 0.23929926753044128, "rewards/total_composite/mean": 0.17074038088321686, "rewards/total_composite/std": 0.10653632879257202, "reward": 0.17074038088321686, "reward_std": 0.10653632134199142, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02135242149233818, "sampling/sampling_logp_difference/max": 5.103124141693115, "sampling/importance_sampling_ratio/min": 0.006077729165554047, "sampling/importance_sampling_ratio/mean": 0.9987974166870117, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07645870675332844, "clip_ratio/low_mean": 0.003349777136463672, "clip_ratio/low_min": 0.003349777136463672, "clip_ratio/high_mean": 0.008041649358347058, "clip_ratio/high_max": 0.008041649358347058, "clip_ratio/region_mean": 0.01139142649481073, "reward_total_mean": 0.17074038088321686, "reward_meter_mean": 0.9907213449478149, "reward_meter_std": 0.006963523104786873, "reward_count_adherence_mean": 0.3571428656578064, "reward_count_adherence_std": 0.07636035233736038, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.45148104429244995, "reward_repeat_penalty_std": 0.23929926753044128, "reward_total_composite_mean": 0.17074038088321686, "reward_total_composite_std": 0.10653632879257202} {"timestamp_utc": "2026-04-11T22:54:14Z", "mode": "train", "global_step": 774, "epoch": 0.031088082901554404, "loss": -0.0134, "grad_norm": 6.2788872718811035, "learning_rate": 7.657575757575757e-06, "num_tokens": 1711058.0, "completions/mean_length": 69.125, "completions/min_length": 65.0, "completions/max_length": 75.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 69.125, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 75.0, "rewards/meter/mean": 0.7258920669555664, "rewards/meter/std": 0.40674278140068054, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7258920669555664, "rewards/total_composite/std": 0.40674278140068054, "reward": 0.7258920669555664, "reward_std": 0.40674278140068054, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.05154965817928314, "sampling/sampling_logp_difference/max": 2.2386436462402344, "sampling/importance_sampling_ratio/min": 0.1066029891371727, "sampling/importance_sampling_ratio/mean": 1.0082036256790161, "sampling/importance_sampling_ratio/max": 1.882003903388977, "entropy": 0.37627044692635536, "clip_ratio/low_mean": 0.005654420121572912, "clip_ratio/low_min": 0.005654420121572912, "clip_ratio/high_mean": 0.022313665016554296, "clip_ratio/high_max": 0.022313665016554296, "clip_ratio/region_mean": 0.027968085138127208, "reward_total_mean": 0.7258920669555664, "reward_meter_mean": 0.7258920669555664, "reward_meter_std": 0.40674278140068054, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7258920669555664, "reward_total_composite_std": 0.40674278140068054} {"timestamp_utc": "2026-04-11T22:54:20Z", "mode": "train", "global_step": 775, "epoch": 0.031128248383339358, "loss": 0.0636, "grad_norm": 2.807114601135254, "learning_rate": 7.654545454545456e-06, "num_tokens": 1713733.0, "completions/mean_length": 168.375, "completions/min_length": 147.0, "completions/max_length": 188.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 168.375, "completions/min_terminated_length": 147.0, "completions/max_terminated_length": 188.0, "rewards/meter/mean": 0.9743660688400269, "rewards/meter/std": 0.04519447684288025, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.10690449178218842, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6060605645179749, "rewards/repeat_penalty/std": 0.13551926612854004, "rewards/total_composite/mean": 0.5430476665496826, "rewards/total_composite/std": 0.16461656987667084, "reward": 0.5430476665496826, "reward_std": 0.16461656987667084, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026703810319304466, "sampling/sampling_logp_difference/max": 12.626859664916992, "sampling/importance_sampling_ratio/min": 3.2826496862981003e-06, "sampling/importance_sampling_ratio/mean": 0.9989210367202759, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06733021000400186, "clip_ratio/low_mean": 0.00418766331858933, "clip_ratio/low_min": 0.00418766331858933, "clip_ratio/high_mean": 0.007139897206798196, "clip_ratio/high_max": 0.007139897206798196, "clip_ratio/region_mean": 0.011327560525387526, "reward_total_mean": 0.5430476665496826, "reward_meter_mean": 0.9743660688400269, "reward_meter_std": 0.04519447684288025, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.10690449178218842, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6060605645179749, "reward_repeat_penalty_std": 0.13551926612854004, "reward_total_composite_mean": 0.5430476665496826, "reward_total_composite_std": 0.16461656987667084} {"timestamp_utc": "2026-04-11T22:54:25Z", "mode": "train", "global_step": 776, "epoch": 0.03116841386512431, "loss": 0.0012, "grad_norm": 7.4227614402771, "learning_rate": 7.651515151515152e-06, "num_tokens": 1715409.0, "completions/mean_length": 54.5, "completions/min_length": 53.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.5, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.9904471635818481, "rewards/meter/std": 0.002613567281514406, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9904471635818481, "rewards/total_composite/std": 0.002613567281514406, "reward": 0.9904471635818481, "reward_std": 0.002613575430586934, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024185476824641228, "sampling/sampling_logp_difference/max": 1.0824594497680664, "sampling/importance_sampling_ratio/min": 0.3387613296508789, "sampling/importance_sampling_ratio/mean": 0.9967302083969116, "sampling/importance_sampling_ratio/max": 1.4630638360977173, "entropy": 0.09682174911722541, "clip_ratio/low_mean": 0.009263938991352916, "clip_ratio/low_min": 0.009263938991352916, "clip_ratio/high_mean": 0.009050324792042375, "clip_ratio/high_max": 0.009050324792042375, "clip_ratio/region_mean": 0.01831426378339529, "reward_total_mean": 0.9904471635818481, "reward_meter_mean": 0.9904471635818481, "reward_meter_std": 0.002613567281514406, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9904471635818481, "reward_total_composite_std": 0.002613567281514406} {"timestamp_utc": "2026-04-11T22:54:29Z", "mode": "train", "global_step": 777, "epoch": 0.031208579346909265, "loss": 0.0046, "grad_norm": 8.215063095092773, "learning_rate": 7.648484848484849e-06, "num_tokens": 1716925.0, "completions/mean_length": 36.5, "completions/min_length": 36.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.5, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.978127121925354, "rewards/meter/std": 0.0063257296569645405, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.978127121925354, "rewards/total_composite/std": 0.0063257296569645405, "reward": 0.978127121925354, "reward_std": 0.006325736176222563, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03158137574791908, "sampling/sampling_logp_difference/max": 1.4631445407867432, "sampling/importance_sampling_ratio/min": 0.23150713741779327, "sampling/importance_sampling_ratio/mean": 1.0036948919296265, "sampling/importance_sampling_ratio/max": 1.6868826150894165, "entropy": 0.11897748988121748, "clip_ratio/low_mean": 0.013795045437291265, "clip_ratio/low_min": 0.013795045437291265, "clip_ratio/high_mean": 0.013795045204460621, "clip_ratio/high_max": 0.013795045204460621, "clip_ratio/region_mean": 0.027590090641751885, "reward_total_mean": 0.978127121925354, "reward_meter_mean": 0.978127121925354, "reward_meter_std": 0.0063257296569645405, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.978127121925354, "reward_total_composite_std": 0.0063257296569645405} {"timestamp_utc": "2026-04-11T22:54:34Z", "mode": "train", "global_step": 778, "epoch": 0.03124874482869422, "loss": 0.014, "grad_norm": 5.06115198135376, "learning_rate": 7.645454545454546e-06, "num_tokens": 1718951.0, "completions/mean_length": 72.25, "completions/min_length": 68.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.25, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9873093366622925, "rewards/meter/std": 0.01036460418254137, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9873093366622925, "rewards/total_composite/std": 0.01036460418254137, "reward": 0.9873093366622925, "reward_std": 0.010364595800638199, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03534568101167679, "sampling/sampling_logp_difference/max": 0.8904938697814941, "sampling/importance_sampling_ratio/min": 0.4104529917240143, "sampling/importance_sampling_ratio/mean": 1.0149582624435425, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.25925541296601295, "clip_ratio/low_mean": 0.02250340231694281, "clip_ratio/low_min": 0.02250340231694281, "clip_ratio/high_mean": 0.020928236190229654, "clip_ratio/high_max": 0.020928236190229654, "clip_ratio/region_mean": 0.043431638507172465, "reward_total_mean": 0.9873093366622925, "reward_meter_mean": 0.9873093366622925, "reward_meter_std": 0.01036460418254137, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9873093366622925, "reward_total_composite_std": 0.01036460418254137} {"timestamp_utc": "2026-04-11T22:54:45Z", "mode": "train", "global_step": 779, "epoch": 0.03128891031047917, "loss": -0.1238, "grad_norm": 1.4244983196258545, "learning_rate": 7.642424242424244e-06, "num_tokens": 1720483.0, "completions/mean_length": 182.5, "completions/min_length": 69.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 72.66667175292969, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 81.0, "rewards/meter/mean": 0.6190794110298157, "rewards/meter/std": 0.3674232065677643, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5616785287857056, "rewards/total_composite/std": 0.43257132172584534, "reward": 0.5616785287857056, "reward_std": 0.43257129192352295, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.07225891947746277, "sampling/sampling_logp_difference/max": 1.2201738357543945, "sampling/importance_sampling_ratio/min": 0.2951788604259491, "sampling/importance_sampling_ratio/mean": 1.0154547691345215, "sampling/importance_sampling_ratio/max": 1.9660524129867554, "entropy": 0.38310878723859787, "clip_ratio/low_mean": 0.008561643771827221, "clip_ratio/low_min": 0.008561643771827221, "clip_ratio/high_mean": 0.030813875840976834, "clip_ratio/high_max": 0.030813875840976834, "clip_ratio/region_mean": 0.039375519612804055, "reward_total_mean": 0.5616785287857056, "reward_meter_mean": 0.6190794110298157, "reward_meter_std": 0.3674232065677643, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5616785287857056, "reward_total_composite_std": 0.43257132172584534} {"timestamp_utc": "2026-04-11T22:54:56Z", "mode": "train", "global_step": 780, "epoch": 0.03132907579226413, "loss": -0.0498, "grad_norm": 3.119354009628296, "learning_rate": 7.639393939393939e-06, "num_tokens": 1722862.0, "completions/mean_length": 174.375, "completions/min_length": 111.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 126.14286041259766, "completions/min_terminated_length": 111.0, "completions/max_terminated_length": 182.0, "rewards/meter/mean": 0.6168656349182129, "rewards/meter/std": 0.4164867103099823, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.7492559552192688, "rewards/repeat_penalty/std": 0.14768067002296448, "rewards/total_composite/mean": 0.4575069546699524, "rewards/total_composite/std": 0.3478102684020996, "reward": 0.4575069546699524, "reward_std": 0.3478102684020996, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03274521231651306, "sampling/sampling_logp_difference/max": 1.078242540359497, "sampling/importance_sampling_ratio/min": 0.34019285440444946, "sampling/importance_sampling_ratio/mean": 1.0024807453155518, "sampling/importance_sampling_ratio/max": 1.7341415882110596, "entropy": 0.24183030799031258, "clip_ratio/low_mean": 0.004807692486792803, "clip_ratio/low_min": 0.004807692486792803, "clip_ratio/high_mean": 0.029140884289518, "clip_ratio/high_max": 0.029140884289518, "clip_ratio/region_mean": 0.0339485767763108, "reward_total_mean": 0.4575069546699524, "reward_meter_mean": 0.6168656349182129, "reward_meter_std": 0.4164867103099823, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.7492559552192688, "reward_repeat_penalty_std": 0.14768067002296448, "reward_total_composite_mean": 0.4575069546699524, "reward_total_composite_std": 0.3478102684020996} {"timestamp_utc": "2026-04-11T22:55:01Z", "mode": "train", "global_step": 781, "epoch": 0.03136924127404908, "loss": -0.0116, "grad_norm": 12.43602180480957, "learning_rate": 7.636363636363638e-06, "num_tokens": 1724537.0, "completions/mean_length": 47.375, "completions/min_length": 44.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.375, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.5619980096817017, "rewards/meter/std": 0.3257886469364166, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.4856577217578888, "rewards/total_composite/std": 0.3285104036331177, "reward": 0.4856577217578888, "reward_std": 0.3285104036331177, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.09212920814752579, "sampling/sampling_logp_difference/max": 1.6957062482833862, "sampling/importance_sampling_ratio/min": 0.18346960842609406, "sampling/importance_sampling_ratio/mean": 0.9970747232437134, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.9118200056254864, "clip_ratio/low_mean": 0.04398810095153749, "clip_ratio/low_min": 0.04398810095153749, "clip_ratio/high_mean": 0.016590908635407686, "clip_ratio/high_max": 0.016590908635407686, "clip_ratio/region_mean": 0.060579009586945176, "reward_total_mean": 0.4856577217578888, "reward_meter_mean": 0.5619980096817017, "reward_meter_std": 0.3257886469364166, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.17251639068126678, "reward_total_composite_mean": 0.4856577217578888, "reward_total_composite_std": 0.3285104036331177} {"timestamp_utc": "2026-04-11T22:55:10Z", "mode": "train", "global_step": 782, "epoch": 0.031409406755834035, "loss": -0.1573, "grad_norm": 1.6532626152038574, "learning_rate": 7.633333333333334e-06, "num_tokens": 1726623.0, "completions/mean_length": 166.75, "completions/min_length": 111.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 117.42857360839844, "completions/min_terminated_length": 111.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.6895759105682373, "rewards/meter/std": 0.26801931858062744, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.824999988079071, "rewards/repeat_penalty/std": 0.19820624589920044, "rewards/total_composite/mean": 0.5303109288215637, "rewards/total_composite/std": 0.28132033348083496, "reward": 0.5303109288215637, "reward_std": 0.28132033348083496, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03518048673868179, "sampling/sampling_logp_difference/max": 1.1620054244995117, "sampling/importance_sampling_ratio/min": 0.31285813450813293, "sampling/importance_sampling_ratio/mean": 1.006204605102539, "sampling/importance_sampling_ratio/max": 1.9478799104690552, "entropy": 0.2663711039349437, "clip_ratio/low_mean": 0.008460594224743545, "clip_ratio/low_min": 0.008460594224743545, "clip_ratio/high_mean": 0.013982121949084103, "clip_ratio/high_max": 0.013982121949084103, "clip_ratio/region_mean": 0.022442716173827648, "reward_total_mean": 0.5303109288215637, "reward_meter_mean": 0.6895759105682373, "reward_meter_std": 0.26801931858062744, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.824999988079071, "reward_repeat_penalty_std": 0.19820624589920044, "reward_total_composite_mean": 0.5303109288215637, "reward_total_composite_std": 0.28132033348083496} {"timestamp_utc": "2026-04-11T22:55:15Z", "mode": "train", "global_step": 783, "epoch": 0.03144957223761899, "loss": 0.0077, "grad_norm": 4.994919776916504, "learning_rate": 7.630303030303031e-06, "num_tokens": 1728611.0, "completions/mean_length": 79.5, "completions/min_length": 76.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 79.5, "completions/min_terminated_length": 76.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.6204426288604736, "rewards/meter/std": 0.3357177972793579, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6204426288604736, "rewards/total_composite/std": 0.3357177972793579, "reward": 0.6204426288604736, "reward_std": 0.3357177972793579, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04477819800376892, "sampling/sampling_logp_difference/max": 1.0638208389282227, "sampling/importance_sampling_ratio/min": 0.34513458609580994, "sampling/importance_sampling_ratio/mean": 1.011016845703125, "sampling/importance_sampling_ratio/max": 1.7397865056991577, "entropy": 0.3720816671848297, "clip_ratio/low_mean": 0.017455301131121814, "clip_ratio/low_min": 0.017455301131121814, "clip_ratio/high_mean": 0.013986280770041049, "clip_ratio/high_max": 0.013986280770041049, "clip_ratio/region_mean": 0.03144158190116286, "reward_total_mean": 0.6204426288604736, "reward_meter_mean": 0.6204426288604736, "reward_meter_std": 0.3357177972793579, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6204426288604736, "reward_total_composite_std": 0.3357177972793579} {"timestamp_utc": "2026-04-11T22:55:20Z", "mode": "train", "global_step": 784, "epoch": 0.03148973771940394, "loss": 0.0052, "grad_norm": 7.485208988189697, "learning_rate": 7.627272727272727e-06, "num_tokens": 1730350.0, "completions/mean_length": 61.375, "completions/min_length": 56.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.375, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.5055776834487915, "rewards/meter/std": 0.32300832867622375, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8333333730697632, "rewards/repeat_penalty/std": 0.17817415297031403, "rewards/total_composite/mean": 0.453016996383667, "rewards/total_composite/std": 0.33306846022605896, "reward": 0.453016996383667, "reward_std": 0.33306846022605896, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.047304026782512665, "sampling/sampling_logp_difference/max": 1.7305419445037842, "sampling/importance_sampling_ratio/min": 0.17718835175037384, "sampling/importance_sampling_ratio/mean": 0.9948754906654358, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.2184693105518818, "clip_ratio/low_mean": 0.018428327050060034, "clip_ratio/low_min": 0.018428327050060034, "clip_ratio/high_mean": 0.012099922401830554, "clip_ratio/high_max": 0.012099922401830554, "clip_ratio/region_mean": 0.030528249451890588, "reward_total_mean": 0.453016996383667, "reward_meter_mean": 0.5055776834487915, "reward_meter_std": 0.32300832867622375, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8333333730697632, "reward_repeat_penalty_std": 0.17817415297031403, "reward_total_composite_mean": 0.453016996383667, "reward_total_composite_std": 0.33306846022605896} {"timestamp_utc": "2026-04-11T22:55:25Z", "mode": "train", "global_step": 785, "epoch": 0.0315299032011889, "loss": 0.0036, "grad_norm": 6.059953212738037, "learning_rate": 7.6242424242424254e-06, "num_tokens": 1732524.0, "completions/mean_length": 104.75, "completions/min_length": 99.0, "completions/max_length": 116.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 104.75, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 116.0, "rewards/meter/mean": 0.49319469928741455, "rewards/meter/std": 0.2711309790611267, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8999999761581421, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.46458256244659424, "rewards/total_composite/std": 0.2836271822452545, "reward": 0.46458256244659424, "reward_std": 0.2836271822452545, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.042036473751068115, "sampling/sampling_logp_difference/max": 1.2848865985870361, "sampling/importance_sampling_ratio/min": 0.27668195962905884, "sampling/importance_sampling_ratio/mean": 1.0086406469345093, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.31474705785512924, "clip_ratio/low_mean": 0.010869022691622376, "clip_ratio/low_min": 0.010869022691622376, "clip_ratio/high_mean": 0.01767290150746703, "clip_ratio/high_max": 0.01767290150746703, "clip_ratio/region_mean": 0.028541924199089408, "reward_total_mean": 0.46458256244659424, "reward_meter_mean": 0.49319469928741455, "reward_meter_std": 0.2711309790611267, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8999999761581421, "reward_repeat_penalty_std": 0.10690449178218842, "reward_total_composite_mean": 0.46458256244659424, "reward_total_composite_std": 0.2836271822452545} {"timestamp_utc": "2026-04-11T22:55:34Z", "mode": "train", "global_step": 786, "epoch": 0.03157006868297385, "loss": -0.0567, "grad_norm": 2.024092435836792, "learning_rate": 7.621212121212122e-06, "num_tokens": 1737190.0, "completions/mean_length": 368.25, "completions/min_length": 322.0, "completions/max_length": 406.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 368.25, "completions/min_terminated_length": 322.0, "completions/max_terminated_length": 406.0, "rewards/meter/mean": 0.7014279365539551, "rewards/meter/std": 0.43751856684684753, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5806276798248291, "rewards/repeat_penalty/std": 0.06668182462453842, "rewards/total_composite/mean": 0.3579794764518738, "rewards/total_composite/std": 0.22696760296821594, "reward": 0.3579794764518738, "reward_std": 0.22696760296821594, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012620110996067524, "sampling/sampling_logp_difference/max": 1.491776466369629, "sampling/importance_sampling_ratio/min": 0.2249726504087448, "sampling/importance_sampling_ratio/mean": 1.0009483098983765, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07080868305638433, "clip_ratio/low_mean": 0.004607091657817364, "clip_ratio/low_min": 0.004607091657817364, "clip_ratio/high_mean": 0.0042421949619892985, "clip_ratio/high_max": 0.0042421949619892985, "clip_ratio/region_mean": 0.008849286619806662, "reward_total_mean": 0.3579794764518738, "reward_meter_mean": 0.7014279365539551, "reward_meter_std": 0.43751856684684753, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5806276798248291, "reward_repeat_penalty_std": 0.06668182462453842, "reward_total_composite_mean": 0.3579794764518738, "reward_total_composite_std": 0.22696760296821594} {"timestamp_utc": "2026-04-11T22:55:40Z", "mode": "train", "global_step": 787, "epoch": 0.031610234164758805, "loss": 0.0123, "grad_norm": 3.6839067935943604, "learning_rate": 7.618181818181819e-06, "num_tokens": 1739852.0, "completions/mean_length": 153.75, "completions/min_length": 135.0, "completions/max_length": 162.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 153.75, "completions/min_terminated_length": 135.0, "completions/max_terminated_length": 162.0, "rewards/meter/mean": 0.7929247617721558, "rewards/meter/std": 0.10911522805690765, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6818181872367859, "rewards/repeat_penalty/std": 0.06872082501649857, "rewards/total_composite/mean": 0.435272216796875, "rewards/total_composite/std": 0.09194310009479523, "reward": 0.435272216796875, "reward_std": 0.09194309264421463, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.028897186741232872, "sampling/sampling_logp_difference/max": 1.9267759323120117, "sampling/importance_sampling_ratio/min": 0.14561693370342255, "sampling/importance_sampling_ratio/mean": 0.9994563460350037, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11316746287047863, "clip_ratio/low_mean": 0.01558087719604373, "clip_ratio/low_min": 0.01558087719604373, "clip_ratio/high_mean": 0.007280809339135885, "clip_ratio/high_max": 0.007280809339135885, "clip_ratio/region_mean": 0.022861686535179615, "reward_total_mean": 0.435272216796875, "reward_meter_mean": 0.7929247617721558, "reward_meter_std": 0.10911522805690765, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6818181872367859, "reward_repeat_penalty_std": 0.06872082501649857, "reward_total_composite_mean": 0.435272216796875, "reward_total_composite_std": 0.09194310009479523} {"timestamp_utc": "2026-04-11T22:55:44Z", "mode": "train", "global_step": 788, "epoch": 0.03165039964654376, "loss": -0.0051, "grad_norm": 7.856208324432373, "learning_rate": 7.6151515151515155e-06, "num_tokens": 1741588.0, "completions/mean_length": 58.0, "completions/min_length": 57.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.6469426155090332, "rewards/meter/std": 0.24868278205394745, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6469426155090332, "rewards/total_composite/std": 0.24868278205394745, "reward": 0.6469426155090332, "reward_std": 0.24868276715278625, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022744039073586464, "sampling/sampling_logp_difference/max": 0.6672461032867432, "sampling/importance_sampling_ratio/min": 0.5131197571754456, "sampling/importance_sampling_ratio/mean": 1.0034902095794678, "sampling/importance_sampling_ratio/max": 1.8282591104507446, "entropy": 0.16343743726611137, "clip_ratio/low_mean": 0.012860541231930256, "clip_ratio/low_min": 0.012860541231930256, "clip_ratio/high_mean": 0.01074189692735672, "clip_ratio/high_max": 0.01074189692735672, "clip_ratio/region_mean": 0.023602438159286976, "reward_total_mean": 0.6469426155090332, "reward_meter_mean": 0.6469426155090332, "reward_meter_std": 0.24868278205394745, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6469426155090332, "reward_total_composite_std": 0.24868278205394745} {"timestamp_utc": "2026-04-11T22:55:49Z", "mode": "train", "global_step": 789, "epoch": 0.03169056512832871, "loss": 0.0131, "grad_norm": 13.922452926635742, "learning_rate": 7.612121212121213e-06, "num_tokens": 1743054.0, "completions/mean_length": 31.25, "completions/min_length": 30.0, "completions/max_length": 32.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.25, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 32.0, "rewards/meter/mean": 0.9813005924224854, "rewards/meter/std": 0.016189413145184517, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9813005924224854, "rewards/total_composite/std": 0.016189413145184517, "reward": 0.9813005924224854, "reward_std": 0.016189415007829666, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.033597491681575775, "sampling/sampling_logp_difference/max": 1.0770900249481201, "sampling/importance_sampling_ratio/min": 0.3405851721763611, "sampling/importance_sampling_ratio/mean": 0.9976152181625366, "sampling/importance_sampling_ratio/max": 1.8372166156768799, "entropy": 0.17746192403137684, "clip_ratio/low_mean": 0.008072916883975267, "clip_ratio/low_min": 0.008072916883975267, "clip_ratio/high_mean": 0.008072916883975267, "clip_ratio/high_max": 0.008072916883975267, "clip_ratio/region_mean": 0.016145833767950535, "reward_total_mean": 0.9813005924224854, "reward_meter_mean": 0.9813005924224854, "reward_meter_std": 0.016189413145184517, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9813005924224854, "reward_total_composite_std": 0.016189413145184517} {"timestamp_utc": "2026-04-11T22:55:54Z", "mode": "train", "global_step": 790, "epoch": 0.03173073061011367, "loss": -0.0039, "grad_norm": 6.977390766143799, "learning_rate": 7.609090909090909e-06, "num_tokens": 1745187.0, "completions/mean_length": 89.625, "completions/min_length": 85.0, "completions/max_length": 92.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.625, "completions/min_terminated_length": 85.0, "completions/max_terminated_length": 92.0, "rewards/meter/mean": 0.8486406207084656, "rewards/meter/std": 0.2562169134616852, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6789125204086304, "rewards/total_composite/std": 0.20497353374958038, "reward": 0.6789125204086304, "reward_std": 0.20497353374958038, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021279064938426018, "sampling/sampling_logp_difference/max": 0.814723014831543, "sampling/importance_sampling_ratio/min": 0.44276192784309387, "sampling/importance_sampling_ratio/mean": 1.0061084032058716, "sampling/importance_sampling_ratio/max": 1.8353419303894043, "entropy": 0.11805877834558487, "clip_ratio/low_mean": 0.009868005756288767, "clip_ratio/low_min": 0.009868005756288767, "clip_ratio/high_mean": 0.019598963437601924, "clip_ratio/high_max": 0.019598963437601924, "clip_ratio/region_mean": 0.02946696919389069, "reward_total_mean": 0.6789125204086304, "reward_meter_mean": 0.8486406207084656, "reward_meter_std": 0.2562169134616852, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6789125204086304, "reward_total_composite_std": 0.20497353374958038} {"timestamp_utc": "2026-04-11T22:56:01Z", "mode": "train", "global_step": 791, "epoch": 0.03177089609189862, "loss": -0.0395, "grad_norm": 1.8626375198364258, "learning_rate": 7.606060606060606e-06, "num_tokens": 1748773.0, "completions/mean_length": 247.25, "completions/min_length": 228.0, "completions/max_length": 258.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 247.25, "completions/min_terminated_length": 228.0, "completions/max_terminated_length": 258.0, "rewards/meter/mean": 0.9959322214126587, "rewards/meter/std": 0.004204806871712208, "rewards/count_adherence/mean": 0.910714328289032, "rewards/count_adherence/std": 0.07393559068441391, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.49047619104385376, "rewards/repeat_penalty/std": 0.17105023562908173, "rewards/total_composite/mean": 0.4338645935058594, "rewards/total_composite/std": 0.12500374019145966, "reward": 0.4338645935058594, "reward_std": 0.12500372529029846, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014479896053671837, "sampling/sampling_logp_difference/max": 2.833479881286621, "sampling/importance_sampling_ratio/min": 0.0588078573346138, "sampling/importance_sampling_ratio/mean": 1.0004253387451172, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.042171002831310034, "clip_ratio/low_mean": 0.0032260402804240584, "clip_ratio/low_min": 0.0032260402804240584, "clip_ratio/high_mean": 0.00779381615575403, "clip_ratio/high_max": 0.00779381615575403, "clip_ratio/region_mean": 0.011019856436178088, "reward_total_mean": 0.4338645935058594, "reward_meter_mean": 0.9959322214126587, "reward_meter_std": 0.004204806871712208, "reward_count_adherence_mean": 0.910714328289032, "reward_count_adherence_std": 0.07393559068441391, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.49047619104385376, "reward_repeat_penalty_std": 0.17105023562908173, "reward_total_composite_mean": 0.4338645935058594, "reward_total_composite_std": 0.12500374019145966} {"timestamp_utc": "2026-04-11T22:56:06Z", "mode": "train", "global_step": 792, "epoch": 0.031811061573683574, "loss": 0.023, "grad_norm": 3.4738333225250244, "learning_rate": 7.603030303030303e-06, "num_tokens": 1751485.0, "completions/mean_length": 135.0, "completions/min_length": 124.0, "completions/max_length": 148.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 135.0, "completions/min_terminated_length": 124.0, "completions/max_terminated_length": 148.0, "rewards/meter/mean": 0.991753101348877, "rewards/meter/std": 0.002629074966534972, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6666666865348816, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6611686944961548, "rewards/total_composite/std": 0.0017527303425595164, "reward": 0.6611686944961548, "reward_std": 0.0017527244053781033, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019971484318375587, "sampling/sampling_logp_difference/max": 2.0009312629699707, "sampling/importance_sampling_ratio/min": 0.13520930707454681, "sampling/importance_sampling_ratio/mean": 1.0068343877792358, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0634374669753015, "clip_ratio/low_mean": 0.010955466306768358, "clip_ratio/low_min": 0.010955466306768358, "clip_ratio/high_mean": 0.00658258656039834, "clip_ratio/high_max": 0.00658258656039834, "clip_ratio/region_mean": 0.017538052867166698, "reward_total_mean": 0.6611686944961548, "reward_meter_mean": 0.991753101348877, "reward_meter_std": 0.002629074966534972, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6666666865348816, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6611686944961548, "reward_total_composite_std": 0.0017527303425595164} {"timestamp_utc": "2026-04-11T22:56:11Z", "mode": "train", "global_step": 793, "epoch": 0.03185122705546853, "loss": 0.0066, "grad_norm": 5.889811038970947, "learning_rate": 7.600000000000001e-06, "num_tokens": 1753253.0, "completions/mean_length": 56.0, "completions/min_length": 51.0, "completions/max_length": 60.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.0, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 60.0, "rewards/meter/mean": 0.9871419668197632, "rewards/meter/std": 0.005153529345989227, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9166666865348816, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.9050641655921936, "rewards/total_composite/std": 0.15341757237911224, "reward": 0.9050641655921936, "reward_std": 0.15341757237911224, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.054076679050922394, "sampling/sampling_logp_difference/max": 2.979457139968872, "sampling/importance_sampling_ratio/min": 0.05082041397690773, "sampling/importance_sampling_ratio/mean": 0.9988625645637512, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15502143744379282, "clip_ratio/low_mean": 0.006658692145720124, "clip_ratio/low_min": 0.006658692145720124, "clip_ratio/high_mean": 0.0335467669647187, "clip_ratio/high_max": 0.0335467669647187, "clip_ratio/region_mean": 0.040205459110438824, "reward_total_mean": 0.9050641655921936, "reward_meter_mean": 0.9871419668197632, "reward_meter_std": 0.005153529345989227, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9166666865348816, "reward_repeat_penalty_std": 0.15430334210395813, "reward_total_composite_mean": 0.9050641655921936, "reward_total_composite_std": 0.15341757237911224} {"timestamp_utc": "2026-04-11T22:56:16Z", "mode": "train", "global_step": 794, "epoch": 0.03189139253725348, "loss": 0.002, "grad_norm": 6.565411567687988, "learning_rate": 7.596969696969697e-06, "num_tokens": 1755042.0, "completions/mean_length": 71.625, "completions/min_length": 69.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.625, "completions/min_terminated_length": 69.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9123142957687378, "rewards/meter/std": 0.09913700073957443, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9123142957687378, "rewards/total_composite/std": 0.09913700073957443, "reward": 0.9123142957687378, "reward_std": 0.09913701564073563, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.034922052174806595, "sampling/sampling_logp_difference/max": 1.3885717391967773, "sampling/importance_sampling_ratio/min": 0.249431312084198, "sampling/importance_sampling_ratio/mean": 1.0043187141418457, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15924948174506426, "clip_ratio/low_mean": 0.019052310031838715, "clip_ratio/low_min": 0.019052310031838715, "clip_ratio/high_mean": 0.007044379832223058, "clip_ratio/high_max": 0.007044379832223058, "clip_ratio/region_mean": 0.026096689864061773, "reward_total_mean": 0.9123142957687378, "reward_meter_mean": 0.9123142957687378, "reward_meter_std": 0.09913700073957443, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9123142957687378, "reward_total_composite_std": 0.09913700073957443} {"timestamp_utc": "2026-04-11T22:56:21Z", "mode": "train", "global_step": 795, "epoch": 0.031931558019038436, "loss": 0.0046, "grad_norm": 2.6046133041381836, "learning_rate": 7.593939393939395e-06, "num_tokens": 1756866.0, "completions/mean_length": 73.0, "completions/min_length": 73.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.0, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.613559365272522, "rewards/meter/std": 0.046057041734457016, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.613559365272522, "rewards/total_composite/std": 0.046057041734457016, "reward": 0.613559365272522, "reward_std": 0.04605703800916672, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01630816049873829, "sampling/sampling_logp_difference/max": 0.9375072717666626, "sampling/importance_sampling_ratio/min": 0.5393000841140747, "sampling/importance_sampling_ratio/mean": 1.0046606063842773, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07319271843880415, "clip_ratio/low_mean": 0.008561643771827221, "clip_ratio/low_min": 0.008561643771827221, "clip_ratio/high_mean": 0.0034246575087308884, "clip_ratio/high_max": 0.0034246575087308884, "clip_ratio/region_mean": 0.01198630128055811, "reward_total_mean": 0.613559365272522, "reward_meter_mean": 0.613559365272522, "reward_meter_std": 0.046057041734457016, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.613559365272522, "reward_total_composite_std": 0.046057041734457016} {"timestamp_utc": "2026-04-11T22:56:28Z", "mode": "train", "global_step": 796, "epoch": 0.03197172350082339, "loss": 0.0182, "grad_norm": 1.5290881395339966, "learning_rate": 7.590909090909091e-06, "num_tokens": 1760371.0, "completions/mean_length": 250.125, "completions/min_length": 248.0, "completions/max_length": 264.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 250.125, "completions/min_terminated_length": 248.0, "completions/max_terminated_length": 264.0, "rewards/meter/mean": 0.6284428834915161, "rewards/meter/std": 0.1288241744041443, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6098901033401489, "rewards/repeat_penalty/std": 0.015540807507932186, "rewards/total_composite/mean": 0.3849101662635803, "rewards/total_composite/std": 0.08409524708986282, "reward": 0.3849101662635803, "reward_std": 0.08409524708986282, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006375753786414862, "sampling/sampling_logp_difference/max": 1.0706424713134766, "sampling/importance_sampling_ratio/min": 0.3427882194519043, "sampling/importance_sampling_ratio/mean": 1.0002148151397705, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.023278776556253433, "clip_ratio/low_mean": 0.0004734848625957966, "clip_ratio/low_min": 0.0004734848625957966, "clip_ratio/high_mean": 0.002516112755984068, "clip_ratio/high_max": 0.002516112755984068, "clip_ratio/region_mean": 0.0029895976185798645, "reward_total_mean": 0.3849101662635803, "reward_meter_mean": 0.6284428834915161, "reward_meter_std": 0.1288241744041443, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6098901033401489, "reward_repeat_penalty_std": 0.015540807507932186, "reward_total_composite_mean": 0.3849101662635803, "reward_total_composite_std": 0.08409524708986282} {"timestamp_utc": "2026-04-11T22:56:33Z", "mode": "train", "global_step": 797, "epoch": 0.032011888982608344, "loss": -0.0121, "grad_norm": 8.541836738586426, "learning_rate": 7.587878787878788e-06, "num_tokens": 1762455.0, "completions/mean_length": 77.5, "completions/min_length": 72.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.5, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.9952297210693359, "rewards/meter/std": 0.002311403863132, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.891382098197937, "rewards/total_composite/std": 0.1964196413755417, "reward": 0.891382098197937, "reward_std": 0.1964196413755417, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.034574542194604874, "sampling/sampling_logp_difference/max": 3.3448140621185303, "sampling/importance_sampling_ratio/min": 0.03526677191257477, "sampling/importance_sampling_ratio/mean": 0.9954510927200317, "sampling/importance_sampling_ratio/max": 1.482370376586914, "entropy": 0.14471599273383617, "clip_ratio/low_mean": 0.003246753243729472, "clip_ratio/low_min": 0.003246753243729472, "clip_ratio/high_mean": 0.009473072132095695, "clip_ratio/high_max": 0.009473072132095695, "clip_ratio/region_mean": 0.012719825375825167, "reward_total_mean": 0.891382098197937, "reward_meter_mean": 0.9952297210693359, "reward_meter_std": 0.002311403863132, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.891382098197937, "reward_total_composite_std": 0.1964196413755417} {"timestamp_utc": "2026-04-11T22:56:38Z", "mode": "train", "global_step": 798, "epoch": 0.0320520544643933, "loss": 0.0167, "grad_norm": 6.7140374183654785, "learning_rate": 7.584848484848486e-06, "num_tokens": 1764291.0, "completions/mean_length": 66.5, "completions/min_length": 64.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.5, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.6813795566558838, "rewards/meter/std": 0.04148668423295021, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6813795566558838, "rewards/total_composite/std": 0.04148668423295021, "reward": 0.6813795566558838, "reward_std": 0.04148669168353081, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023791270330548286, "sampling/sampling_logp_difference/max": 0.9566974639892578, "sampling/importance_sampling_ratio/min": 0.5285674333572388, "sampling/importance_sampling_ratio/mean": 1.010377049446106, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1323620891198516, "clip_ratio/low_mean": 0.020607191254384816, "clip_ratio/low_min": 0.020607191254384816, "clip_ratio/high_mean": 0.00390625, "clip_ratio/high_max": 0.00390625, "clip_ratio/region_mean": 0.024513441254384816, "reward_total_mean": 0.6813795566558838, "reward_meter_mean": 0.6813795566558838, "reward_meter_std": 0.04148668423295021, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6813795566558838, "reward_total_composite_std": 0.04148668423295021} {"timestamp_utc": "2026-04-11T22:56:45Z", "mode": "train", "global_step": 799, "epoch": 0.03209221994617825, "loss": 0.0044, "grad_norm": 1.2270196676254272, "learning_rate": 7.581818181818183e-06, "num_tokens": 1767790.0, "completions/mean_length": 239.375, "completions/min_length": 237.0, "completions/max_length": 240.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 239.375, "completions/min_terminated_length": 237.0, "completions/max_terminated_length": 240.0, "rewards/meter/mean": 0.7462765574455261, "rewards/meter/std": 0.09833642095327377, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6000000238418579, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.38379937410354614, "rewards/total_composite/std": 0.05057300627231598, "reward": 0.38379937410354614, "reward_std": 0.050573013722896576, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007426128257066011, "sampling/sampling_logp_difference/max": 2.6020090579986572, "sampling/importance_sampling_ratio/min": 0.07412450760602951, "sampling/importance_sampling_ratio/mean": 0.9996190071105957, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.030161422211676836, "clip_ratio/low_mean": 0.003645833523478359, "clip_ratio/low_min": 0.003645833523478359, "clip_ratio/high_mean": 0.002109704539179802, "clip_ratio/high_max": 0.002109704539179802, "clip_ratio/region_mean": 0.005755538062658161, "reward_total_mean": 0.38379937410354614, "reward_meter_mean": 0.7462765574455261, "reward_meter_std": 0.09833642095327377, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6000000238418579, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.38379937410354614, "reward_total_composite_std": 0.05057300627231598} {"timestamp_utc": "2026-04-11T22:56:50Z", "mode": "train", "global_step": 800, "epoch": 0.032132385427963206, "loss": 0.0432, "grad_norm": 4.79632568359375, "learning_rate": 7.57878787878788e-06, "num_tokens": 1770025.0, "completions/mean_length": 119.375, "completions/min_length": 110.0, "completions/max_length": 130.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 119.375, "completions/min_terminated_length": 110.0, "completions/max_terminated_length": 130.0, "rewards/meter/mean": 0.5170607566833496, "rewards/meter/std": 0.48065322637557983, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.4120614528656006, "rewards/total_composite/std": 0.386055052280426, "reward": 0.4120614528656006, "reward_std": 0.386055052280426, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04103344678878784, "sampling/sampling_logp_difference/max": 1.1491317749023438, "sampling/importance_sampling_ratio/min": 0.3169117867946625, "sampling/importance_sampling_ratio/mean": 0.9977750182151794, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.23464529775083065, "clip_ratio/low_mean": 0.024260351667180657, "clip_ratio/low_min": 0.024260351667180657, "clip_ratio/high_mean": 0.0143592240056023, "clip_ratio/high_max": 0.0143592240056023, "clip_ratio/region_mean": 0.03861957567278296, "reward_total_mean": 0.4120614528656006, "reward_meter_mean": 0.5170607566833496, "reward_meter_std": 0.48065322637557983, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.10690449178218842, "reward_total_composite_mean": 0.4120614528656006, "reward_total_composite_std": 0.386055052280426} {"timestamp_utc": "2026-04-11T22:58:16Z", "mode": "eval", "global_step": 800, "epoch": 0.032132385427963206, "eval_loss": NaN, "eval_runtime": 85.9157, "eval_samples_per_second": 1.21, "eval_steps_per_second": 0.151, "eval_num_tokens": 1770025.0, "eval_completions/mean_length": 234.29807692307693, "eval_completions/min_length": 63.84615384615385, "eval_completions/max_length": 461.9230769230769, "eval_completions/clipped_ratio": 0.057692307692307696, "eval_completions/mean_terminated_length": 216.9835216815655, "eval_completions/min_terminated_length": 63.84615384615385, "eval_completions/max_terminated_length": 414.46153846153845, "eval_rewards/meter/mean": 0.6419616112342248, "eval_rewards/meter/std": 0.3754527878302794, "eval_rewards/count_adherence/mean": 0.9552615697567279, "eval_rewards/count_adherence/std": 0.08013723160211857, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/repeat_penalty/mean": 0.6556147245260385, "eval_rewards/repeat_penalty/std": 0.24277657327743676, "eval_rewards/total_composite/mean": 0.3991968219096844, "eval_rewards/total_composite/std": 0.3098094039238416, "eval_reward": 0.3991968219096844, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.0054089168552309275, "eval_sampling/sampling_logp_difference/max": 0.7075341939926147, "eval_sampling/importance_sampling_ratio/min": 0.5230464408030877, "eval_sampling/importance_sampling_ratio/mean": 1.0009051194557776, "eval_sampling/importance_sampling_ratio/max": 1.2961730773632343, "eval_entropy": 0.047992275741237864, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.3991968219096844, "eval_reward_meter_mean": 0.6419616112342248, "eval_reward_meter_std": 0.3754527878302794, "eval_reward_count_adherence_mean": 0.9552615697567279, "eval_reward_count_adherence_std": 0.08013723160211857, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_repeat_penalty_mean": 0.6556147245260385, "eval_reward_repeat_penalty_std": 0.24277657327743676, "eval_reward_total_composite_mean": 0.3991968219096844, "eval_reward_total_composite_std": 0.3098094039238416} {"timestamp_utc": "2026-04-11T22:58:23Z", "mode": "train", "global_step": 801, "epoch": 0.03217255090974816, "loss": 0.0018, "grad_norm": 4.925754070281982, "learning_rate": 7.5757575757575764e-06, "num_tokens": 1771810.0, "completions/mean_length": 67.125, "completions/min_length": 67.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.125, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.7430859804153442, "rewards/meter/std": 0.04014641046524048, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7430859804153442, "rewards/total_composite/std": 0.04014641046524048, "reward": 0.7430859804153442, "reward_std": 0.040146395564079285, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02094501443207264, "sampling/sampling_logp_difference/max": 1.098017692565918, "sampling/importance_sampling_ratio/min": 0.33353158831596375, "sampling/importance_sampling_ratio/mean": 1.0070921182632446, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.08063888642936945, "clip_ratio/low_mean": 0.018601843621581793, "clip_ratio/low_min": 0.018601843621581793, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.018601843621581793, "reward_total_mean": 0.7430859804153442, "reward_meter_mean": 0.7430859804153442, "reward_meter_std": 0.04014641046524048, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7430859804153442, "reward_total_composite_std": 0.04014641046524048} {"timestamp_utc": "2026-04-11T22:58:28Z", "mode": "train", "global_step": 802, "epoch": 0.032212716391533114, "loss": 0.0172, "grad_norm": 7.339356422424316, "learning_rate": 7.572727272727274e-06, "num_tokens": 1773359.0, "completions/mean_length": 37.625, "completions/min_length": 36.0, "completions/max_length": 40.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.625, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 40.0, "rewards/meter/mean": 0.13143374025821686, "rewards/meter/std": 0.11004206538200378, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.13143374025821686, "rewards/total_composite/std": 0.11004206538200378, "reward": 0.13143374025821686, "reward_std": 0.11004206538200378, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03102065436542034, "sampling/sampling_logp_difference/max": 1.9875779151916504, "sampling/importance_sampling_ratio/min": 0.13702690601348877, "sampling/importance_sampling_ratio/mean": 0.9923664331436157, "sampling/importance_sampling_ratio/max": 1.671715497970581, "entropy": 0.1240494973026216, "clip_ratio/low_mean": 0.009703947464004159, "clip_ratio/low_min": 0.009703947464004159, "clip_ratio/high_mean": 0.00995732587762177, "clip_ratio/high_max": 0.00995732587762177, "clip_ratio/region_mean": 0.01966127334162593, "reward_total_mean": 0.13143374025821686, "reward_meter_mean": 0.13143374025821686, "reward_meter_std": 0.11004206538200378, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.13143374025821686, "reward_total_composite_std": 0.11004206538200378} {"timestamp_utc": "2026-04-11T22:58:32Z", "mode": "train", "global_step": 803, "epoch": 0.03225288187331807, "loss": 0.0049, "grad_norm": 4.46676778793335, "learning_rate": 7.56969696969697e-06, "num_tokens": 1775325.0, "completions/mean_length": 70.75, "completions/min_length": 70.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 70.75, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9594069123268127, "rewards/meter/std": 0.05366376414895058, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9594069123268127, "rewards/total_composite/std": 0.05366376414895058, "reward": 0.9594069123268127, "reward_std": 0.05366375669836998, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03665493428707123, "sampling/sampling_logp_difference/max": 1.9053394794464111, "sampling/importance_sampling_ratio/min": 0.14877213537693024, "sampling/importance_sampling_ratio/mean": 0.9985708594322205, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.1510819010436535, "clip_ratio/low_mean": 0.005281690042465925, "clip_ratio/low_min": 0.005281690042465925, "clip_ratio/high_mean": 0.026509752846322954, "clip_ratio/high_max": 0.026509752846322954, "clip_ratio/region_mean": 0.03179144288878888, "reward_total_mean": 0.9594069123268127, "reward_meter_mean": 0.9594069123268127, "reward_meter_std": 0.05366376414895058, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9594069123268127, "reward_total_composite_std": 0.05366376414895058} {"timestamp_utc": "2026-04-11T22:58:44Z", "mode": "train", "global_step": 804, "epoch": 0.03229304735510302, "loss": 0.0506, "grad_norm": 0.5559709072113037, "learning_rate": 7.566666666666667e-06, "num_tokens": 1780007.0, "completions/mean_length": 505.25, "completions/min_length": 493.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 503.0, "completions/min_terminated_length": 493.0, "completions/max_terminated_length": 508.0, "rewards/meter/mean": 0.9569629430770874, "rewards/meter/std": 0.102637879550457, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.0235702246427536, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.4376780688762665, "rewards/repeat_penalty/std": 0.22492921352386475, "rewards/total_composite/mean": 0.38479870557785034, "rewards/total_composite/std": 0.20677605271339417, "reward": 0.38479870557785034, "reward_std": 0.20677603781223297, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.006570629775524139, "sampling/sampling_logp_difference/max": 2.7531256675720215, "sampling/importance_sampling_ratio/min": 0.06372835487127304, "sampling/importance_sampling_ratio/mean": 1.0004603862762451, "sampling/importance_sampling_ratio/max": 1.9659916162490845, "entropy": 0.014583299867808819, "clip_ratio/low_mean": 0.0004926113178953528, "clip_ratio/low_min": 0.0004926113178953528, "clip_ratio/high_mean": 0.003243225917685777, "clip_ratio/high_max": 0.003243225917685777, "clip_ratio/region_mean": 0.00373583723558113, "reward_total_mean": 0.38479870557785034, "reward_meter_mean": 0.9569629430770874, "reward_meter_std": 0.102637879550457, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.0235702246427536, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.4376780688762665, "reward_repeat_penalty_std": 0.22492921352386475, "reward_total_composite_mean": 0.38479870557785034, "reward_total_composite_std": 0.20677605271339417} {"timestamp_utc": "2026-04-11T22:58:48Z", "mode": "train", "global_step": 805, "epoch": 0.032333212836887976, "loss": 0.0024, "grad_norm": 4.009762287139893, "learning_rate": 7.563636363636364e-06, "num_tokens": 1781817.0, "completions/mean_length": 67.25, "completions/min_length": 67.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 67.25, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.7574520111083984, "rewards/meter/std": 0.03009505569934845, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7574520111083984, "rewards/total_composite/std": 0.03009505569934845, "reward": 0.7574520111083984, "reward_std": 0.0300950538367033, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020206134766340256, "sampling/sampling_logp_difference/max": 1.227433443069458, "sampling/importance_sampling_ratio/min": 0.29304373264312744, "sampling/importance_sampling_ratio/mean": 0.9979568719863892, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.05657489877194166, "clip_ratio/low_mean": 0.011085874866694212, "clip_ratio/low_min": 0.011085874866694212, "clip_ratio/high_mean": 0.00932835799176246, "clip_ratio/high_max": 0.00932835799176246, "clip_ratio/region_mean": 0.02041423285845667, "reward_total_mean": 0.7574520111083984, "reward_meter_mean": 0.7574520111083984, "reward_meter_std": 0.03009505569934845, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7574520111083984, "reward_total_composite_std": 0.03009505569934845} {"timestamp_utc": "2026-04-11T22:58:53Z", "mode": "train", "global_step": 806, "epoch": 0.03237337831867293, "loss": 0.0021, "grad_norm": 3.6810977458953857, "learning_rate": 7.560606060606062e-06, "num_tokens": 1783594.0, "completions/mean_length": 58.125, "completions/min_length": 57.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.125, "completions/min_terminated_length": 57.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.990125298500061, "rewards/meter/std": 0.0014512698398903012, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8333333730697632, "rewards/repeat_penalty/std": 0.17817415297031403, "rewards/total_composite/mean": 0.8252000212669373, "rewards/total_composite/std": 0.17693018913269043, "reward": 0.8252000212669373, "reward_std": 0.17693018913269043, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.023440338671207428, "sampling/sampling_logp_difference/max": 1.1792564392089844, "sampling/importance_sampling_ratio/min": 0.30750730633735657, "sampling/importance_sampling_ratio/mean": 0.9979320168495178, "sampling/importance_sampling_ratio/max": 1.3977869749069214, "entropy": 0.11172830406576395, "clip_ratio/low_mean": 0.004310344811528921, "clip_ratio/low_min": 0.004310344811528921, "clip_ratio/high_mean": 0.01062974869273603, "clip_ratio/high_max": 0.01062974869273603, "clip_ratio/region_mean": 0.01494009350426495, "reward_total_mean": 0.8252000212669373, "reward_meter_mean": 0.990125298500061, "reward_meter_std": 0.0014512698398903012, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8333333730697632, "reward_repeat_penalty_std": 0.17817415297031403, "reward_total_composite_mean": 0.8252000212669373, "reward_total_composite_std": 0.17693018913269043} {"timestamp_utc": "2026-04-11T22:58:58Z", "mode": "train", "global_step": 807, "epoch": 0.032413543800457884, "loss": -0.0084, "grad_norm": 5.640214920043945, "learning_rate": 7.557575757575758e-06, "num_tokens": 1785577.0, "completions/mean_length": 75.875, "completions/min_length": 72.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.875, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.7918260097503662, "rewards/meter/std": 0.3541868031024933, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7303093671798706, "rewards/total_composite/std": 0.35869720578193665, "reward": 0.7303093671798706, "reward_std": 0.35869717597961426, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02571277692914009, "sampling/sampling_logp_difference/max": 1.1890771389007568, "sampling/importance_sampling_ratio/min": 0.4486599266529083, "sampling/importance_sampling_ratio/mean": 1.0031059980392456, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.12689008563756943, "clip_ratio/low_mean": 0.00854225957300514, "clip_ratio/low_min": 0.00854225957300514, "clip_ratio/high_mean": 0.008098726975731552, "clip_ratio/high_max": 0.008098726975731552, "clip_ratio/region_mean": 0.01664098654873669, "reward_total_mean": 0.7303093671798706, "reward_meter_mean": 0.7918260097503662, "reward_meter_std": 0.3541868031024933, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7303093671798706, "reward_total_composite_std": 0.35869720578193665} {"timestamp_utc": "2026-04-11T22:59:02Z", "mode": "train", "global_step": 808, "epoch": 0.03245370928224284, "loss": -0.0018, "grad_norm": 14.179214477539062, "learning_rate": 7.5545454545454555e-06, "num_tokens": 1787137.0, "completions/mean_length": 38.0, "completions/min_length": 38.0, "completions/max_length": 38.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.0, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 38.0, "rewards/meter/mean": 0.8494601845741272, "rewards/meter/std": 0.3245563507080078, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8494601845741272, "rewards/total_composite/std": 0.3245563507080078, "reward": 0.8494601845741272, "reward_std": 0.3245563209056854, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.026692412793636322, "sampling/sampling_logp_difference/max": 2.183236598968506, "sampling/importance_sampling_ratio/min": 0.11267625540494919, "sampling/importance_sampling_ratio/mean": 0.9977914690971375, "sampling/importance_sampling_ratio/max": 1.5366783142089844, "entropy": 0.09128655772656202, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.016447368543595076, "clip_ratio/high_max": 0.016447368543595076, "clip_ratio/region_mean": 0.016447368543595076, "reward_total_mean": 0.8494601845741272, "reward_meter_mean": 0.8494601845741272, "reward_meter_std": 0.3245563507080078, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8494601845741272, "reward_total_composite_std": 0.3245563507080078} {"timestamp_utc": "2026-04-11T22:59:09Z", "mode": "train", "global_step": 809, "epoch": 0.03249387476402779, "loss": -0.0031, "grad_norm": 1.1023591756820679, "learning_rate": 7.551515151515152e-06, "num_tokens": 1790539.0, "completions/mean_length": 218.25, "completions/min_length": 206.0, "completions/max_length": 230.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 218.25, "completions/min_terminated_length": 206.0, "completions/max_terminated_length": 230.0, "rewards/meter/mean": 0.9922986030578613, "rewards/meter/std": 0.004149852320551872, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5681818723678589, "rewards/repeat_penalty/std": 0.1735115498304367, "rewards/total_composite/mean": 0.5638895034790039, "rewards/total_composite/std": 0.17219732701778412, "reward": 0.5638895034790039, "reward_std": 0.17219732701778412, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011151120997965336, "sampling/sampling_logp_difference/max": 1.5731306076049805, "sampling/importance_sampling_ratio/min": 0.20739488303661346, "sampling/importance_sampling_ratio/mean": 1.0026495456695557, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03606141824275255, "clip_ratio/low_mean": 0.005849295761436224, "clip_ratio/low_min": 0.005849295761436224, "clip_ratio/high_mean": 0.003968499368056655, "clip_ratio/high_max": 0.003968499368056655, "clip_ratio/region_mean": 0.009817795129492879, "reward_total_mean": 0.5638895034790039, "reward_meter_mean": 0.9922986030578613, "reward_meter_std": 0.004149852320551872, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5681818723678589, "reward_repeat_penalty_std": 0.1735115498304367, "reward_total_composite_mean": 0.5638895034790039, "reward_total_composite_std": 0.17219732701778412} {"timestamp_utc": "2026-04-11T22:59:14Z", "mode": "train", "global_step": 810, "epoch": 0.032534040245812745, "loss": 0.0017, "grad_norm": 5.80548095703125, "learning_rate": 7.548484848484849e-06, "num_tokens": 1792366.0, "completions/mean_length": 71.375, "completions/min_length": 70.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.375, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.9742876887321472, "rewards/meter/std": 0.029914017766714096, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9332020878791809, "rewards/total_composite/std": 0.11529982835054398, "reward": 0.9332020878791809, "reward_std": 0.11529984325170517, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0373886339366436, "sampling/sampling_logp_difference/max": 1.9366556406021118, "sampling/importance_sampling_ratio/min": 0.14418534934520721, "sampling/importance_sampling_ratio/mean": 1.0060287714004517, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15044390503317118, "clip_ratio/low_mean": 0.005306840990670025, "clip_ratio/low_min": 0.005306840990670025, "clip_ratio/high_mean": 0.03143186215311289, "clip_ratio/high_max": 0.03143186215311289, "clip_ratio/region_mean": 0.036738703143782914, "reward_total_mean": 0.9332020878791809, "reward_meter_mean": 0.9742876887321472, "reward_meter_std": 0.029914017766714096, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.9332020878791809, "reward_total_composite_std": 0.11529982835054398} {"timestamp_utc": "2026-04-11T22:59:20Z", "mode": "train", "global_step": 811, "epoch": 0.0325742057275977, "loss": 0.0192, "grad_norm": 2.172598123550415, "learning_rate": 7.545454545454546e-06, "num_tokens": 1795876.0, "completions/mean_length": 203.75, "completions/min_length": 196.0, "completions/max_length": 229.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 203.75, "completions/min_terminated_length": 196.0, "completions/max_terminated_length": 229.0, "rewards/meter/mean": 0.710443377494812, "rewards/meter/std": 0.2666179835796356, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.4431818127632141, "rewards/repeat_penalty/std": 0.22498852014541626, "rewards/total_composite/mean": 0.3340369462966919, "rewards/total_composite/std": 0.2214316725730896, "reward": 0.3340369462966919, "reward_std": 0.2214316576719284, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021875398233532906, "sampling/sampling_logp_difference/max": 4.384680271148682, "sampling/importance_sampling_ratio/min": 0.012466873973608017, "sampling/importance_sampling_ratio/mean": 1.0012760162353516, "sampling/importance_sampling_ratio/max": 1.9448559284210205, "entropy": 0.1404771413654089, "clip_ratio/low_mean": 0.006675369921140373, "clip_ratio/low_min": 0.006675369921140373, "clip_ratio/high_mean": 0.011296228156425059, "clip_ratio/high_max": 0.011296228156425059, "clip_ratio/region_mean": 0.01797159807756543, "reward_total_mean": 0.3340369462966919, "reward_meter_mean": 0.710443377494812, "reward_meter_std": 0.2666179835796356, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.4431818127632141, "reward_repeat_penalty_std": 0.22498852014541626, "reward_total_composite_mean": 0.3340369462966919, "reward_total_composite_std": 0.2214316725730896} {"timestamp_utc": "2026-04-11T22:59:25Z", "mode": "train", "global_step": 812, "epoch": 0.03261437120938265, "loss": 0.0139, "grad_norm": 8.286386489868164, "learning_rate": 7.542424242424244e-06, "num_tokens": 1797566.0, "completions/mean_length": 65.25, "completions/min_length": 65.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.25, "completions/min_terminated_length": 65.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9978810548782349, "rewards/meter/std": 0.0005103643052279949, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9563552141189575, "rewards/total_composite/std": 0.11796265840530396, "reward": 0.9563552141189575, "reward_std": 0.11796264350414276, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007206229493021965, "sampling/sampling_logp_difference/max": 0.5880947113037109, "sampling/importance_sampling_ratio/min": 0.5553844571113586, "sampling/importance_sampling_ratio/mean": 1.002901315689087, "sampling/importance_sampling_ratio/max": 1.3499990701675415, "entropy": 0.036765412194654346, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.005769230774603784, "clip_ratio/high_max": 0.005769230774603784, "clip_ratio/region_mean": 0.005769230774603784, "reward_total_mean": 0.9563552141189575, "reward_meter_mean": 0.9978810548782349, "reward_meter_std": 0.0005103643052279949, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.9563552141189575, "reward_total_composite_std": 0.11796265840530396} {"timestamp_utc": "2026-04-11T22:59:34Z", "mode": "train", "global_step": 813, "epoch": 0.03265453669116761, "loss": 0.0079, "grad_norm": 1.0310794115066528, "learning_rate": 7.53939393939394e-06, "num_tokens": 1801761.0, "completions/mean_length": 332.375, "completions/min_length": 302.0, "completions/max_length": 353.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 332.375, "completions/min_terminated_length": 302.0, "completions/max_terminated_length": 353.0, "rewards/meter/mean": 0.9938352704048157, "rewards/meter/std": 0.008550022728741169, "rewards/count_adherence/mean": 0.890625, "rewards/count_adherence/std": 0.04419417306780815, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5850183963775635, "rewards/repeat_penalty/std": 0.009098809212446213, "rewards/total_composite/mean": 0.517345666885376, "rewards/total_composite/std": 0.012124452739953995, "reward": 0.517345666885376, "reward_std": 0.012124458327889442, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0052232746966183186, "sampling/sampling_logp_difference/max": 0.7512289881706238, "sampling/importance_sampling_ratio/min": 0.47178637981414795, "sampling/importance_sampling_ratio/mean": 1.0009092092514038, "sampling/importance_sampling_ratio/max": 1.6740483045578003, "entropy": 0.02348946128040552, "clip_ratio/low_mean": 0.005218350415816531, "clip_ratio/low_min": 0.005218350415816531, "clip_ratio/high_mean": 0.0019113150192424655, "clip_ratio/high_max": 0.0019113150192424655, "clip_ratio/region_mean": 0.007129665435058996, "reward_total_mean": 0.517345666885376, "reward_meter_mean": 0.9938352704048157, "reward_meter_std": 0.008550022728741169, "reward_count_adherence_mean": 0.890625, "reward_count_adherence_std": 0.04419417306780815, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5850183963775635, "reward_repeat_penalty_std": 0.009098809212446213, "reward_total_composite_mean": 0.517345666885376, "reward_total_composite_std": 0.012124452739953995} {"timestamp_utc": "2026-04-11T22:59:40Z", "mode": "train", "global_step": 814, "epoch": 0.03269470217295257, "loss": -0.001, "grad_norm": 13.524036407470703, "learning_rate": 7.536363636363637e-06, "num_tokens": 1804516.0, "completions/mean_length": 159.375, "completions/min_length": 158.0, "completions/max_length": 167.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 159.375, "completions/min_terminated_length": 158.0, "completions/max_terminated_length": 167.0, "rewards/meter/mean": 0.9979825019836426, "rewards/meter/std": 0.0006693408940918744, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6666666865348816, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6653216481208801, "rewards/total_composite/std": 0.00044622019049711525, "reward": 0.6653216481208801, "reward_std": 0.00044622053974308074, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.005735776387155056, "sampling/sampling_logp_difference/max": 2.232571601867676, "sampling/importance_sampling_ratio/min": 0.10725226998329163, "sampling/importance_sampling_ratio/mean": 0.9996488094329834, "sampling/importance_sampling_ratio/max": 1.493585467338562, "entropy": 0.03128327080048621, "clip_ratio/low_mean": 0.00237341778120026, "clip_ratio/low_min": 0.00237341778120026, "clip_ratio/high_mean": 0.00237341778120026, "clip_ratio/high_max": 0.00237341778120026, "clip_ratio/region_mean": 0.00474683556240052, "reward_total_mean": 0.6653216481208801, "reward_meter_mean": 0.9979825019836426, "reward_meter_std": 0.0006693408940918744, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6666666865348816, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6653216481208801, "reward_total_composite_std": 0.00044622019049711525} {"timestamp_utc": "2026-04-11T22:59:44Z", "mode": "train", "global_step": 815, "epoch": 0.03273486765473752, "loss": 0.0458, "grad_norm": 8.952130317687988, "learning_rate": 7.533333333333334e-06, "num_tokens": 1806317.0, "completions/mean_length": 62.125, "completions/min_length": 54.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.125, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.9308251142501831, "rewards/meter/std": 0.15261007845401764, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.8894620537757874, "rewards/total_composite/std": 0.17642506957054138, "reward": 0.8894620537757874, "reward_std": 0.1764250546693802, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024382825940847397, "sampling/sampling_logp_difference/max": 1.424285888671875, "sampling/importance_sampling_ratio/min": 0.2406802922487259, "sampling/importance_sampling_ratio/mean": 0.9998781085014343, "sampling/importance_sampling_ratio/max": 1.5671416521072388, "entropy": 0.10703426506370306, "clip_ratio/low_mean": 0.011092530796304345, "clip_ratio/low_min": 0.011092530796304345, "clip_ratio/high_mean": 0.016365812392905354, "clip_ratio/high_max": 0.016365812392905354, "clip_ratio/region_mean": 0.0274583431892097, "reward_total_mean": 0.8894620537757874, "reward_meter_mean": 0.9308251142501831, "reward_meter_std": 0.15261007845401764, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.8894620537757874, "reward_total_composite_std": 0.17642506957054138} {"timestamp_utc": "2026-04-11T22:59:49Z", "mode": "train", "global_step": 816, "epoch": 0.032775033136522476, "loss": -0.0066, "grad_norm": 8.190361022949219, "learning_rate": 7.530303030303031e-06, "num_tokens": 1807821.0, "completions/mean_length": 39.0, "completions/min_length": 38.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 39.0, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.7446458339691162, "rewards/meter/std": 0.30041757225990295, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7446458339691162, "rewards/total_composite/std": 0.30041757225990295, "reward": 0.7446458339691162, "reward_std": 0.30041757225990295, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.059971921145915985, "sampling/sampling_logp_difference/max": 2.114863395690918, "sampling/importance_sampling_ratio/min": 0.12064976990222931, "sampling/importance_sampling_ratio/mean": 0.9998385310173035, "sampling/importance_sampling_ratio/max": 1.8297884464263916, "entropy": 0.16819044947624207, "clip_ratio/low_mean": 0.023026316426694393, "clip_ratio/low_min": 0.023026316426694393, "clip_ratio/high_mean": 0.028449730249121785, "clip_ratio/high_max": 0.028449730249121785, "clip_ratio/region_mean": 0.05147604667581618, "reward_total_mean": 0.7446458339691162, "reward_meter_mean": 0.7446458339691162, "reward_meter_std": 0.30041757225990295, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7446458339691162, "reward_total_composite_std": 0.30041757225990295} {"timestamp_utc": "2026-04-11T22:59:54Z", "mode": "train", "global_step": 817, "epoch": 0.03281519861830743, "loss": 0.0413, "grad_norm": 3.7302210330963135, "learning_rate": 7.5272727272727274e-06, "num_tokens": 1810070.0, "completions/mean_length": 107.125, "completions/min_length": 102.0, "completions/max_length": 115.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 107.125, "completions/min_terminated_length": 102.0, "completions/max_terminated_length": 115.0, "rewards/meter/mean": 0.3294169008731842, "rewards/meter/std": 0.28731057047843933, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7750000357627869, "rewards/repeat_penalty/std": 0.12817399203777313, "rewards/total_composite/mean": 0.2755756676197052, "rewards/total_composite/std": 0.2410556823015213, "reward": 0.2755756676197052, "reward_std": 0.2410556823015213, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01802036352455616, "sampling/sampling_logp_difference/max": 1.4399476051330566, "sampling/importance_sampling_ratio/min": 0.2369401901960373, "sampling/importance_sampling_ratio/mean": 1.0044699907302856, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.05588802928104997, "clip_ratio/low_mean": 0.01120058260858059, "clip_ratio/low_min": 0.01120058260858059, "clip_ratio/high_mean": 0.013432800536975265, "clip_ratio/high_max": 0.013432800536975265, "clip_ratio/region_mean": 0.024633383145555854, "reward_total_mean": 0.2755756676197052, "reward_meter_mean": 0.3294169008731842, "reward_meter_std": 0.28731057047843933, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7750000357627869, "reward_repeat_penalty_std": 0.12817399203777313, "reward_total_composite_mean": 0.2755756676197052, "reward_total_composite_std": 0.2410556823015213} {"timestamp_utc": "2026-04-11T23:00:03Z", "mode": "train", "global_step": 818, "epoch": 0.032855364100092384, "loss": 0.0108, "grad_norm": 1.0928229093551636, "learning_rate": 7.524242424242425e-06, "num_tokens": 1814452.0, "completions/mean_length": 357.75, "completions/min_length": 346.0, "completions/max_length": 384.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 357.75, "completions/min_terminated_length": 346.0, "completions/max_terminated_length": 384.0, "rewards/meter/mean": 0.9965507984161377, "rewards/meter/std": 0.0037117046304047108, "rewards/count_adherence/mean": 0.9583333134651184, "rewards/count_adherence/std": 0.05750546231865883, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5523655414581299, "rewards/repeat_penalty/std": 0.10810358822345734, "rewards/total_composite/mean": 0.5283541083335876, "rewards/total_composite/std": 0.11482103914022446, "reward": 0.5283541083335876, "reward_std": 0.11482104659080505, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00884958729147911, "sampling/sampling_logp_difference/max": 1.062759518623352, "sampling/importance_sampling_ratio/min": 0.3455010652542114, "sampling/importance_sampling_ratio/mean": 1.0001471042633057, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03675817488692701, "clip_ratio/low_mean": 0.003796361561398953, "clip_ratio/low_min": 0.003796361561398953, "clip_ratio/high_mean": 0.006020042230375111, "clip_ratio/high_max": 0.006020042230375111, "clip_ratio/region_mean": 0.009816403791774064, "reward_total_mean": 0.5283541083335876, "reward_meter_mean": 0.9965507984161377, "reward_meter_std": 0.0037117046304047108, "reward_count_adherence_mean": 0.9583333134651184, "reward_count_adherence_std": 0.05750546231865883, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5523655414581299, "reward_repeat_penalty_std": 0.10810358822345734, "reward_total_composite_mean": 0.5283541083335876, "reward_total_composite_std": 0.11482103914022446} {"timestamp_utc": "2026-04-11T23:00:08Z", "mode": "train", "global_step": 819, "epoch": 0.03289552958187734, "loss": 0.0279, "grad_norm": 6.570438861846924, "learning_rate": 7.521212121212121e-06, "num_tokens": 1816193.0, "completions/mean_length": 61.625, "completions/min_length": 59.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.625, "completions/min_terminated_length": 59.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.9746619462966919, "rewards/meter/std": 0.020627282559871674, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9746619462966919, "rewards/total_composite/std": 0.020627282559871674, "reward": 0.9746619462966919, "reward_std": 0.020627308636903763, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.02879003807902336, "sampling/sampling_logp_difference/max": 1.3429758548736572, "sampling/importance_sampling_ratio/min": 0.26106762886047363, "sampling/importance_sampling_ratio/mean": 0.9997745156288147, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10824673250317574, "clip_ratio/low_mean": 0.013770792167633772, "clip_ratio/low_min": 0.013770792167633772, "clip_ratio/high_mean": 0.022682114504277706, "clip_ratio/high_max": 0.022682114504277706, "clip_ratio/region_mean": 0.03645290667191148, "reward_total_mean": 0.9746619462966919, "reward_meter_mean": 0.9746619462966919, "reward_meter_std": 0.020627282559871674, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9746619462966919, "reward_total_composite_std": 0.020627282559871674} {"timestamp_utc": "2026-04-11T23:00:14Z", "mode": "train", "global_step": 820, "epoch": 0.03293569506366229, "loss": 0.0185, "grad_norm": 2.3457443714141846, "learning_rate": 7.518181818181819e-06, "num_tokens": 1818812.0, "completions/mean_length": 149.375, "completions/min_length": 143.0, "completions/max_length": 156.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 149.375, "completions/min_terminated_length": 143.0, "completions/max_terminated_length": 156.0, "rewards/meter/mean": 0.7805307507514954, "rewards/meter/std": 0.1253480166196823, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5714285373687744, "rewards/repeat_penalty/std": 0.17074695229530334, "rewards/total_composite/mean": 0.4300089478492737, "rewards/total_composite/std": 0.0887664407491684, "reward": 0.4300089478492737, "reward_std": 0.0887664407491684, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015637392178177834, "sampling/sampling_logp_difference/max": 1.2027087211608887, "sampling/importance_sampling_ratio/min": 0.3003794550895691, "sampling/importance_sampling_ratio/mean": 1.0010299682617188, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.06724543264135718, "clip_ratio/low_mean": 0.007311595429200679, "clip_ratio/low_min": 0.007311595429200679, "clip_ratio/high_mean": 0.008506174897775054, "clip_ratio/high_max": 0.008506174897775054, "clip_ratio/region_mean": 0.015817770326975733, "reward_total_mean": 0.4300089478492737, "reward_meter_mean": 0.7805307507514954, "reward_meter_std": 0.1253480166196823, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5714285373687744, "reward_repeat_penalty_std": 0.17074695229530334, "reward_total_composite_mean": 0.4300089478492737, "reward_total_composite_std": 0.0887664407491684} {"timestamp_utc": "2026-04-11T23:00:23Z", "mode": "train", "global_step": 821, "epoch": 0.032975860545447246, "loss": -0.1471, "grad_norm": 0.9305291175842285, "learning_rate": 7.515151515151516e-06, "num_tokens": 1820662.0, "completions/mean_length": 190.25, "completions/min_length": 77.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.25, "completions/mean_terminated_length": 83.0, "completions/min_terminated_length": 77.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.5668963193893433, "rewards/meter/std": 0.39650967717170715, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.4629100561141968, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5668963193893433, "rewards/total_composite/std": 0.39650967717170715, "reward": 0.5668963193893433, "reward_std": 0.39650964736938477, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019788384437561035, "sampling/sampling_logp_difference/max": 0.7466448545455933, "sampling/importance_sampling_ratio/min": 0.6215139627456665, "sampling/importance_sampling_ratio/mean": 1.0027402639389038, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.07134009897708893, "clip_ratio/low_mean": 0.007159017724916339, "clip_ratio/low_min": 0.007159017724916339, "clip_ratio/high_mean": 0.01558885129634291, "clip_ratio/high_max": 0.01558885129634291, "clip_ratio/region_mean": 0.022747869021259248, "reward_total_mean": 0.5668963193893433, "reward_meter_mean": 0.5668963193893433, "reward_meter_std": 0.39650967717170715, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.4629100561141968, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5668963193893433, "reward_total_composite_std": 0.39650967717170715} {"timestamp_utc": "2026-04-11T23:00:28Z", "mode": "train", "global_step": 822, "epoch": 0.0330160260272322, "loss": 0.0105, "grad_norm": 3.704124689102173, "learning_rate": 7.512121212121213e-06, "num_tokens": 1822669.0, "completions/mean_length": 77.875, "completions/min_length": 75.0, "completions/max_length": 82.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 77.875, "completions/min_terminated_length": 75.0, "completions/max_terminated_length": 82.0, "rewards/meter/mean": 0.8737984299659729, "rewards/meter/std": 0.14763900637626648, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8737984299659729, "rewards/total_composite/std": 0.14763900637626648, "reward": 0.8737984299659729, "reward_std": 0.14763899147510529, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.018405567854642868, "sampling/sampling_logp_difference/max": 1.2704942226409912, "sampling/importance_sampling_ratio/min": 0.28069284558296204, "sampling/importance_sampling_ratio/mean": 1.0058021545410156, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.09962678607553244, "clip_ratio/low_mean": 0.006382113788276911, "clip_ratio/low_min": 0.006382113788276911, "clip_ratio/high_mean": 0.009740259731188416, "clip_ratio/high_max": 0.009740259731188416, "clip_ratio/region_mean": 0.016122373519465327, "reward_total_mean": 0.8737984299659729, "reward_meter_mean": 0.8737984299659729, "reward_meter_std": 0.14763900637626648, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8737984299659729, "reward_total_composite_std": 0.14763900637626648} {"timestamp_utc": "2026-04-11T23:00:35Z", "mode": "train", "global_step": 823, "epoch": 0.033056191509017153, "loss": 0.0055, "grad_norm": 2.836785078048706, "learning_rate": 7.509090909090909e-06, "num_tokens": 1826360.0, "completions/mean_length": 284.375, "completions/min_length": 272.0, "completions/max_length": 297.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 284.375, "completions/min_terminated_length": 272.0, "completions/max_terminated_length": 297.0, "rewards/meter/mean": 0.9983997344970703, "rewards/meter/std": 0.0007759786094538867, "rewards/count_adherence/mean": 0.9821428656578064, "rewards/count_adherence/std": 0.05050762742757797, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5952796936035156, "rewards/repeat_penalty/std": 0.08133196830749512, "rewards/total_composite/mean": 0.5862284898757935, "rewards/total_composite/std": 0.09859947860240936, "reward": 0.5862284898757935, "reward_std": 0.09859946370124817, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015353661961853504, "sampling/sampling_logp_difference/max": 2.6265687942504883, "sampling/importance_sampling_ratio/min": 0.07232620567083359, "sampling/importance_sampling_ratio/mean": 1.0005059242248535, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03474967950023711, "clip_ratio/low_mean": 0.002257478976389393, "clip_ratio/low_min": 0.002257478976389393, "clip_ratio/high_mean": 0.0065078792395070195, "clip_ratio/high_max": 0.0065078792395070195, "clip_ratio/region_mean": 0.008765358215896413, "reward_total_mean": 0.5862284898757935, "reward_meter_mean": 0.9983997344970703, "reward_meter_std": 0.0007759786094538867, "reward_count_adherence_mean": 0.9821428656578064, "reward_count_adherence_std": 0.05050762742757797, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5952796936035156, "reward_repeat_penalty_std": 0.08133196830749512, "reward_total_composite_mean": 0.5862284898757935, "reward_total_composite_std": 0.09859947860240936} {"timestamp_utc": "2026-04-11T23:00:40Z", "mode": "train", "global_step": 824, "epoch": 0.03309635699080211, "loss": 0.0168, "grad_norm": 4.631432056427002, "learning_rate": 7.5060606060606065e-06, "num_tokens": 1828232.0, "completions/mean_length": 71.0, "completions/min_length": 70.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.0, "completions/min_terminated_length": 70.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.7449524998664856, "rewards/meter/std": 0.04608132317662239, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7449524998664856, "rewards/total_composite/std": 0.04608132317662239, "reward": 0.7449524998664856, "reward_std": 0.04608132690191269, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01667422614991665, "sampling/sampling_logp_difference/max": 2.1282317638397217, "sampling/importance_sampling_ratio/min": 0.1190476045012474, "sampling/importance_sampling_ratio/mean": 0.9986546039581299, "sampling/importance_sampling_ratio/max": 1.426641583442688, "entropy": 0.04505776287987828, "clip_ratio/low_mean": 0.0017361111240461469, "clip_ratio/low_min": 0.0017361111240461469, "clip_ratio/high_mean": 0.0070436508394777775, "clip_ratio/high_max": 0.0070436508394777775, "clip_ratio/region_mean": 0.008779761963523924, "reward_total_mean": 0.7449524998664856, "reward_meter_mean": 0.7449524998664856, "reward_meter_std": 0.04608132317662239, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7449524998664856, "reward_total_composite_std": 0.04608132317662239} {"timestamp_utc": "2026-04-11T23:00:45Z", "mode": "train", "global_step": 825, "epoch": 0.03313652247258706, "loss": 0.0481, "grad_norm": 4.750690937042236, "learning_rate": 7.503030303030303e-06, "num_tokens": 1830123.0, "completions/mean_length": 68.375, "completions/min_length": 64.0, "completions/max_length": 74.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.375, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 74.0, "rewards/meter/mean": 0.4837471544742584, "rewards/meter/std": 0.22915734350681305, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4837471544742584, "rewards/total_composite/std": 0.22915734350681305, "reward": 0.4837471544742584, "reward_std": 0.22915734350681305, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.031915146857500076, "sampling/sampling_logp_difference/max": 1.8651485443115234, "sampling/importance_sampling_ratio/min": 0.15487320721149445, "sampling/importance_sampling_ratio/mean": 0.9999799728393555, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10288840066641569, "clip_ratio/low_mean": 0.024154700804501772, "clip_ratio/low_min": 0.024154700804501772, "clip_ratio/high_mean": 0.028171515092253685, "clip_ratio/high_max": 0.028171515092253685, "clip_ratio/region_mean": 0.05232621589675546, "reward_total_mean": 0.4837471544742584, "reward_meter_mean": 0.4837471544742584, "reward_meter_std": 0.22915734350681305, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.4837471544742584, "reward_total_composite_std": 0.22915734350681305} {"timestamp_utc": "2026-04-11T23:00:54Z", "mode": "train", "global_step": 826, "epoch": 0.033176687954372015, "loss": 0.0159, "grad_norm": 1.1773563623428345, "learning_rate": 7.500000000000001e-06, "num_tokens": 1835244.0, "completions/mean_length": 400.125, "completions/min_length": 375.0, "completions/max_length": 421.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 400.125, "completions/min_terminated_length": 375.0, "completions/max_terminated_length": 421.0, "rewards/meter/mean": 0.7846977710723877, "rewards/meter/std": 0.39648592472076416, "rewards/count_adherence/mean": 0.9545454978942871, "rewards/count_adherence/std": 0.0485929399728775, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5507364273071289, "rewards/repeat_penalty/std": 0.050370436161756516, "rewards/total_composite/mean": 0.3971686363220215, "rewards/total_composite/std": 0.1935662180185318, "reward": 0.3971686363220215, "reward_std": 0.1935662031173706, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007981406524777412, "sampling/sampling_logp_difference/max": 1.5345345735549927, "sampling/importance_sampling_ratio/min": 0.25283992290496826, "sampling/importance_sampling_ratio/mean": 1.0009510517120361, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0334128841641359, "clip_ratio/low_mean": 0.0033315176842734218, "clip_ratio/low_min": 0.0033315176842734218, "clip_ratio/high_mean": 0.005249909590929747, "clip_ratio/high_max": 0.005249909590929747, "clip_ratio/region_mean": 0.008581427275203168, "reward_total_mean": 0.3971686363220215, "reward_meter_mean": 0.7846977710723877, "reward_meter_std": 0.39648592472076416, "reward_count_adherence_mean": 0.9545454978942871, "reward_count_adherence_std": 0.0485929399728775, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5507364273071289, "reward_repeat_penalty_std": 0.050370436161756516, "reward_total_composite_mean": 0.3971686363220215, "reward_total_composite_std": 0.1935662180185318} {"timestamp_utc": "2026-04-11T23:00:59Z", "mode": "train", "global_step": 827, "epoch": 0.03321685343615697, "loss": -0.0012, "grad_norm": 3.7937092781066895, "learning_rate": 7.496969696969698e-06, "num_tokens": 1837278.0, "completions/mean_length": 68.25, "completions/min_length": 68.0, "completions/max_length": 69.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.25, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 69.0, "rewards/meter/mean": 0.15423895418643951, "rewards/meter/std": 0.07883358746767044, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.15423895418643951, "rewards/total_composite/std": 0.07883358746767044, "reward": 0.15423895418643951, "reward_std": 0.07883358746767044, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.016422174870967865, "sampling/sampling_logp_difference/max": 0.6457552909851074, "sampling/importance_sampling_ratio/min": 0.5242664217948914, "sampling/importance_sampling_ratio/mean": 0.9993041157722473, "sampling/importance_sampling_ratio/max": 1.73274564743042, "entropy": 0.05680654477328062, "clip_ratio/low_mean": 0.009191176504828036, "clip_ratio/low_min": 0.009191176504828036, "clip_ratio/high_mean": 0.005434782709926367, "clip_ratio/high_max": 0.005434782709926367, "clip_ratio/region_mean": 0.014625959214754403, "reward_total_mean": 0.15423895418643951, "reward_meter_mean": 0.15423895418643951, "reward_meter_std": 0.07883358746767044, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.15423895418643951, "reward_total_composite_std": 0.07883358746767044} {"timestamp_utc": "2026-04-11T23:01:04Z", "mode": "train", "global_step": 828, "epoch": 0.03325701891794192, "loss": 0.0145, "grad_norm": 6.0185723304748535, "learning_rate": 7.493939393939395e-06, "num_tokens": 1839058.0, "completions/mean_length": 63.5, "completions/min_length": 61.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 63.5, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9804123640060425, "rewards/meter/std": 0.014910156838595867, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9804123640060425, "rewards/total_composite/std": 0.014910156838595867, "reward": 0.9804123640060425, "reward_std": 0.014910157769918442, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.028718305751681328, "sampling/sampling_logp_difference/max": 1.3754212856292725, "sampling/importance_sampling_ratio/min": 0.25273311138153076, "sampling/importance_sampling_ratio/mean": 1.0006060600280762, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.11623660661280155, "clip_ratio/low_mean": 0.0038470644503831863, "clip_ratio/low_min": 0.0038470644503831863, "clip_ratio/high_mean": 0.01206992007791996, "clip_ratio/high_max": 0.01206992007791996, "clip_ratio/region_mean": 0.015916984528303146, "reward_total_mean": 0.9804123640060425, "reward_meter_mean": 0.9804123640060425, "reward_meter_std": 0.014910156838595867, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9804123640060425, "reward_total_composite_std": 0.014910156838595867} {"timestamp_utc": "2026-04-11T23:01:10Z", "mode": "train", "global_step": 829, "epoch": 0.03329718439972688, "loss": 0.0432, "grad_norm": 1.7107356786727905, "learning_rate": 7.490909090909092e-06, "num_tokens": 1841757.0, "completions/mean_length": 173.375, "completions/min_length": 151.0, "completions/max_length": 186.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 173.375, "completions/min_terminated_length": 151.0, "completions/max_terminated_length": 186.0, "rewards/meter/mean": 0.9961064457893372, "rewards/meter/std": 0.0029631657525897026, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.625, "rewards/repeat_penalty/std": 0.16967642307281494, "rewards/total_composite/mean": 0.6229597926139832, "rewards/total_composite/std": 0.17044374346733093, "reward": 0.6229597926139832, "reward_std": 0.17044374346733093, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017439143732190132, "sampling/sampling_logp_difference/max": 1.495316505432129, "sampling/importance_sampling_ratio/min": 0.22417764365673065, "sampling/importance_sampling_ratio/mean": 0.9979692101478577, "sampling/importance_sampling_ratio/max": 1.7753796577453613, "entropy": 0.05448292032815516, "clip_ratio/low_mean": 0.002016129030380398, "clip_ratio/low_min": 0.002016129030380398, "clip_ratio/high_mean": 0.016328177880495787, "clip_ratio/high_max": 0.016328177880495787, "clip_ratio/region_mean": 0.018344306910876185, "reward_total_mean": 0.6229597926139832, "reward_meter_mean": 0.9961064457893372, "reward_meter_std": 0.0029631657525897026, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.625, "reward_repeat_penalty_std": 0.16967642307281494, "reward_total_composite_mean": 0.6229597926139832, "reward_total_composite_std": 0.17044374346733093} {"timestamp_utc": "2026-04-11T23:01:15Z", "mode": "train", "global_step": 830, "epoch": 0.03333734988151183, "loss": 0.034, "grad_norm": 12.176047325134277, "learning_rate": 7.487878787878788e-06, "num_tokens": 1843710.0, "completions/mean_length": 74.125, "completions/min_length": 68.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.125, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9026094079017639, "rewards/meter/std": 0.14784321188926697, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9026094079017639, "rewards/total_composite/std": 0.14784321188926697, "reward": 0.9026094079017639, "reward_std": 0.14784321188926697, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.060764994472265244, "sampling/sampling_logp_difference/max": 5.8967156410217285, "sampling/importance_sampling_ratio/min": 0.0027484570164233446, "sampling/importance_sampling_ratio/mean": 0.9942511916160583, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.12799056991934776, "clip_ratio/low_mean": 0.007117270142771304, "clip_ratio/low_min": 0.007117270142771304, "clip_ratio/high_mean": 0.019965628627687693, "clip_ratio/high_max": 0.019965628627687693, "clip_ratio/region_mean": 0.027082898770458996, "reward_total_mean": 0.9026094079017639, "reward_meter_mean": 0.9026094079017639, "reward_meter_std": 0.14784321188926697, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9026094079017639, "reward_total_composite_std": 0.14784321188926697} {"timestamp_utc": "2026-04-11T23:01:21Z", "mode": "train", "global_step": 831, "epoch": 0.033377515363296785, "loss": -0.003, "grad_norm": 4.106361389160156, "learning_rate": 7.484848484848486e-06, "num_tokens": 1845975.0, "completions/mean_length": 108.125, "completions/min_length": 105.0, "completions/max_length": 113.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 108.125, "completions/min_terminated_length": 105.0, "completions/max_terminated_length": 113.0, "rewards/meter/mean": 0.3786888122558594, "rewards/meter/std": 0.35290807485580444, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.3143823742866516, "rewards/total_composite/std": 0.282669335603714, "reward": 0.3143823742866516, "reward_std": 0.282669335603714, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.034197013825178146, "sampling/sampling_logp_difference/max": 2.057035207748413, "sampling/importance_sampling_ratio/min": 0.12783241271972656, "sampling/importance_sampling_ratio/mean": 0.9974236488342285, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.12897953018546104, "clip_ratio/low_mean": 0.016522707068361342, "clip_ratio/low_min": 0.016522707068361342, "clip_ratio/high_mean": 0.01371849060524255, "clip_ratio/high_max": 0.01371849060524255, "clip_ratio/region_mean": 0.030241197673603892, "reward_total_mean": 0.3143823742866516, "reward_meter_mean": 0.3786888122558594, "reward_meter_std": 0.35290807485580444, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.1035098284482956, "reward_total_composite_mean": 0.3143823742866516, "reward_total_composite_std": 0.282669335603714} {"timestamp_utc": "2026-04-11T23:01:32Z", "mode": "train", "global_step": 832, "epoch": 0.03341768084508174, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 7.481818181818182e-06, "num_tokens": 1847831.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.6360248923301697, "rewards/meter/std": 0.25572946667671204, "rewards/count_adherence/mean": 0.8602941036224365, "rewards/count_adherence/std": 0.0304440688341856, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5477695465087891, "rewards/repeat_penalty/std": 0.011733362451195717, "rewards/total_composite/mean": 0.3008078336715698, "rewards/total_composite/std": 0.12469936162233353, "reward": 0.3008078336715698, "reward_std": 0.12469936162233353, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.3008078336715698, "reward_meter_mean": 0.6360248923301697, "reward_meter_std": 0.25572946667671204, "reward_count_adherence_mean": 0.8602941036224365, "reward_count_adherence_std": 0.0304440688341856, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5477695465087891, "reward_repeat_penalty_std": 0.011733362451195717, "reward_total_composite_mean": 0.3008078336715698, "reward_total_composite_std": 0.12469936162233353} {"timestamp_utc": "2026-04-11T23:01:36Z", "mode": "train", "global_step": 833, "epoch": 0.03345784632686669, "loss": 0.0048, "grad_norm": 8.433792114257812, "learning_rate": 7.47878787878788e-06, "num_tokens": 1849335.0, "completions/mean_length": 35.0, "completions/min_length": 35.0, "completions/max_length": 35.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.0, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 35.0, "rewards/meter/mean": 0.9924166798591614, "rewards/meter/std": 0.00017309709801338613, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9924166798591614, "rewards/total_composite/std": 0.00017309709801338613, "reward": 0.9924166798591614, "reward_std": 0.00017310312250629067, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008245354518294334, "sampling/sampling_logp_difference/max": 0.6460732221603394, "sampling/importance_sampling_ratio/min": 0.5240997672080994, "sampling/importance_sampling_ratio/mean": 1.0005563497543335, "sampling/importance_sampling_ratio/max": 1.1176583766937256, "entropy": 0.04187649488449097, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9924166798591614, "reward_meter_mean": 0.9924166798591614, "reward_meter_std": 0.00017309709801338613, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9924166798591614, "reward_total_composite_std": 0.00017309709801338613} {"timestamp_utc": "2026-04-11T23:01:41Z", "mode": "train", "global_step": 834, "epoch": 0.03349801180865165, "loss": 0.0342, "grad_norm": 7.6343560218811035, "learning_rate": 7.4757575757575765e-06, "num_tokens": 1851111.0, "completions/mean_length": 68.0, "completions/min_length": 62.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 68.0, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.40049153566360474, "rewards/meter/std": 0.3537753224372864, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.40049153566360474, "rewards/total_composite/std": 0.3537753224372864, "reward": 0.40049153566360474, "reward_std": 0.3537753224372864, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0617922842502594, "sampling/sampling_logp_difference/max": 3.229440689086914, "sampling/importance_sampling_ratio/min": 0.03957962989807129, "sampling/importance_sampling_ratio/mean": 1.0015736818313599, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.33059297781437635, "clip_ratio/low_mean": 0.035882155993022025, "clip_ratio/low_min": 0.035882155993022025, "clip_ratio/high_mean": 0.009836265817284584, "clip_ratio/high_max": 0.009836265817284584, "clip_ratio/region_mean": 0.04571842181030661, "reward_total_mean": 0.40049153566360474, "reward_meter_mean": 0.40049153566360474, "reward_meter_std": 0.3537753224372864, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.40049153566360474, "reward_total_composite_std": 0.3537753224372864} {"timestamp_utc": "2026-04-11T23:01:46Z", "mode": "train", "global_step": 835, "epoch": 0.0335381772904366, "loss": 0.0177, "grad_norm": 1.6731441020965576, "learning_rate": 7.472727272727274e-06, "num_tokens": 1853656.0, "completions/mean_length": 150.125, "completions/min_length": 149.0, "completions/max_length": 157.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 150.125, "completions/min_terminated_length": 149.0, "completions/max_terminated_length": 157.0, "rewards/meter/mean": 0.9238545894622803, "rewards/meter/std": 0.2039637267589569, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6666666865348816, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6159030199050903, "rewards/total_composite/std": 0.13597580790519714, "reward": 0.6159030199050903, "reward_std": 0.13597580790519714, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00628118310123682, "sampling/sampling_logp_difference/max": 1.1030478477478027, "sampling/importance_sampling_ratio/min": 0.3318580687046051, "sampling/importance_sampling_ratio/mean": 0.9989848136901855, "sampling/importance_sampling_ratio/max": 1.6532901525497437, "entropy": 0.03071731375530362, "clip_ratio/low_mean": 0.0007961783558130264, "clip_ratio/low_min": 0.0007961783558130264, "clip_ratio/high_mean": 0.004172259592451155, "clip_ratio/high_max": 0.004172259592451155, "clip_ratio/region_mean": 0.004968437948264182, "reward_total_mean": 0.6159030199050903, "reward_meter_mean": 0.9238545894622803, "reward_meter_std": 0.2039637267589569, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6666666865348816, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.6159030199050903, "reward_total_composite_std": 0.13597580790519714} {"timestamp_utc": "2026-04-11T23:01:52Z", "mode": "train", "global_step": 836, "epoch": 0.033578342772221555, "loss": -0.0007, "grad_norm": 2.8023297786712646, "learning_rate": 7.46969696969697e-06, "num_tokens": 1856194.0, "completions/mean_length": 142.25, "completions/min_length": 139.0, "completions/max_length": 150.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 142.25, "completions/min_terminated_length": 139.0, "completions/max_terminated_length": 150.0, "rewards/meter/mean": 0.8496776819229126, "rewards/meter/std": 0.3326415419578552, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.8214285373687744, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.7131145596504211, "rewards/total_composite/std": 0.2920505106449127, "reward": 0.7131145596504211, "reward_std": 0.29205048084259033, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012105772271752357, "sampling/sampling_logp_difference/max": 1.3084237575531006, "sampling/importance_sampling_ratio/min": 0.2702457010746002, "sampling/importance_sampling_ratio/mean": 0.9984580278396606, "sampling/importance_sampling_ratio/max": 1.5496021509170532, "entropy": 0.044805840123444796, "clip_ratio/low_mean": 0.001773406460415572, "clip_ratio/low_min": 0.001773406460415572, "clip_ratio/high_mean": 0.007069471699651331, "clip_ratio/high_max": 0.007069471699651331, "clip_ratio/region_mean": 0.008842878160066903, "reward_total_mean": 0.7131145596504211, "reward_meter_mean": 0.8496776819229126, "reward_meter_std": 0.3326415419578552, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.8214285373687744, "reward_repeat_penalty_std": 0.06613000482320786, "reward_total_composite_mean": 0.7131145596504211, "reward_total_composite_std": 0.2920505106449127} {"timestamp_utc": "2026-04-11T23:01:56Z", "mode": "train", "global_step": 837, "epoch": 0.03361850825400651, "loss": 0.0053, "grad_norm": 11.809463500976562, "learning_rate": 7.4666666666666675e-06, "num_tokens": 1857618.0, "completions/mean_length": 33.0, "completions/min_length": 33.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 33.0, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9952504634857178, "rewards/meter/std": 0.0004852505517192185, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9952504634857178, "rewards/total_composite/std": 0.0004852505517192185, "reward": 0.9952504634857178, "reward_std": 0.0004852571291849017, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.017783869057893753, "sampling/sampling_logp_difference/max": 0.7500922679901123, "sampling/importance_sampling_ratio/min": 0.4723230004310608, "sampling/importance_sampling_ratio/mean": 1.0051133632659912, "sampling/importance_sampling_ratio/max": 1.5762113332748413, "entropy": 0.0944258663803339, "clip_ratio/low_mean": 0.01515151560306549, "clip_ratio/low_min": 0.01515151560306549, "clip_ratio/high_mean": 0.011363636702299118, "clip_ratio/high_max": 0.011363636702299118, "clip_ratio/region_mean": 0.02651515230536461, "reward_total_mean": 0.9952504634857178, "reward_meter_mean": 0.9952504634857178, "reward_meter_std": 0.0004852505517192185, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9952504634857178, "reward_total_composite_std": 0.0004852505517192185} {"timestamp_utc": "2026-04-11T23:02:04Z", "mode": "train", "global_step": 838, "epoch": 0.03365867373579146, "loss": 0.0061, "grad_norm": 2.168991804122925, "learning_rate": 7.463636363636364e-06, "num_tokens": 1861717.0, "completions/mean_length": 295.375, "completions/min_length": 264.0, "completions/max_length": 331.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 295.375, "completions/min_terminated_length": 264.0, "completions/max_terminated_length": 331.0, "rewards/meter/mean": 0.869155764579773, "rewards/meter/std": 0.35062167048454285, "rewards/count_adherence/mean": 0.828125, "rewards/count_adherence/std": 0.06469365209341049, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 0.5626935362815857, "rewards/repeat_penalty/std": 0.04896574467420578, "rewards/total_composite/mean": 0.4079222083091736, "rewards/total_composite/std": 0.1708296537399292, "reward": 0.4079222083091736, "reward_std": 0.170829638838768, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009027719497680664, "sampling/sampling_logp_difference/max": 1.8500938415527344, "sampling/importance_sampling_ratio/min": 0.15722240507602692, "sampling/importance_sampling_ratio/mean": 0.9985241889953613, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.02827131818048656, "clip_ratio/low_mean": 0.0017137863032985479, "clip_ratio/low_min": 0.0017137863032985479, "clip_ratio/high_mean": 0.004315092111937702, "clip_ratio/high_max": 0.004315092111937702, "clip_ratio/region_mean": 0.0060288784152362496, "reward_total_mean": 0.4079222083091736, "reward_meter_mean": 0.869155764579773, "reward_meter_std": 0.35062167048454285, "reward_count_adherence_mean": 0.828125, "reward_count_adherence_std": 0.06469365209341049, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 0.5626935362815857, "reward_repeat_penalty_std": 0.04896574467420578, "reward_total_composite_mean": 0.4079222083091736, "reward_total_composite_std": 0.1708296537399292} {"timestamp_utc": "2026-04-11T23:02:09Z", "mode": "train", "global_step": 839, "epoch": 0.033698839217576416, "loss": 0.0131, "grad_norm": 2.164283275604248, "learning_rate": 7.460606060606061e-06, "num_tokens": 1863934.0, "completions/mean_length": 112.125, "completions/min_length": 109.0, "completions/max_length": 119.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 112.125, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 119.0, "rewards/meter/mean": 0.9943764209747314, "rewards/meter/std": 0.0010588886216282845, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.795501172542572, "rewards/total_composite/std": 0.0008471080800518394, "reward": 0.795501172542572, "reward_std": 0.0008471118635497987, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008892908692359924, "sampling/sampling_logp_difference/max": 0.7849429249763489, "sampling/importance_sampling_ratio/min": 0.4561457335948944, "sampling/importance_sampling_ratio/mean": 1.0025136470794678, "sampling/importance_sampling_ratio/max": 1.7390761375427246, "entropy": 0.047560357954353094, "clip_ratio/low_mean": 0.006605345057323575, "clip_ratio/low_min": 0.006605345057323575, "clip_ratio/high_mean": 0.0022935778833925724, "clip_ratio/high_max": 0.0022935778833925724, "clip_ratio/region_mean": 0.008898922940716147, "reward_total_mean": 0.795501172542572, "reward_meter_mean": 0.9943764209747314, "reward_meter_std": 0.0010588886216282845, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.795501172542572, "reward_total_composite_std": 0.0008471080800518394} {"timestamp_utc": "2026-04-11T23:02:14Z", "mode": "train", "global_step": 840, "epoch": 0.03373900469936137, "loss": 0.036, "grad_norm": 5.7492289543151855, "learning_rate": 7.4575757575757575e-06, "num_tokens": 1865375.0, "completions/mean_length": 37.125, "completions/min_length": 36.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.125, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.7395036816596985, "rewards/meter/std": 0.2953052818775177, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7395036816596985, "rewards/total_composite/std": 0.2953052818775177, "reward": 0.7395036816596985, "reward_std": 0.2953052818775177, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01924203522503376, "sampling/sampling_logp_difference/max": 0.40451645851135254, "sampling/importance_sampling_ratio/min": 0.6779606938362122, "sampling/importance_sampling_ratio/mean": 1.0085185766220093, "sampling/importance_sampling_ratio/max": 1.4985777139663696, "entropy": 0.11148730758577585, "clip_ratio/low_mean": 0.006410256493836641, "clip_ratio/low_min": 0.006410256493836641, "clip_ratio/high_mean": 0.010228979168459773, "clip_ratio/high_max": 0.010228979168459773, "clip_ratio/region_mean": 0.016639235662296414, "reward_total_mean": 0.7395036816596985, "reward_meter_mean": 0.7395036816596985, "reward_meter_std": 0.2953052818775177, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.7395036816596985, "reward_total_composite_std": 0.2953052818775177} {"timestamp_utc": "2026-04-11T23:02:19Z", "mode": "train", "global_step": 841, "epoch": 0.033779170181146324, "loss": -0.0141, "grad_norm": 3.028710126876831, "learning_rate": 7.454545454545456e-06, "num_tokens": 1867245.0, "completions/mean_length": 64.75, "completions/min_length": 64.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.75, "completions/min_terminated_length": 64.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.9528952240943909, "rewards/meter/std": 0.01815449446439743, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9528952240943909, "rewards/total_composite/std": 0.01815449446439743, "reward": 0.9528952240943909, "reward_std": 0.018154479563236237, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.01176523882895708, "sampling/sampling_logp_difference/max": 0.8644721508026123, "sampling/importance_sampling_ratio/min": 0.5180370807647705, "sampling/importance_sampling_ratio/mean": 1.0040351152420044, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.045040544122457504, "clip_ratio/low_mean": 0.00390625, "clip_ratio/low_min": 0.00390625, "clip_ratio/high_mean": 0.0018656715983524919, "clip_ratio/high_max": 0.0018656715983524919, "clip_ratio/region_mean": 0.005771921598352492, "reward_total_mean": 0.9528952240943909, "reward_meter_mean": 0.9528952240943909, "reward_meter_std": 0.01815449446439743, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.9528952240943909, "reward_total_composite_std": 0.01815449446439743} {"timestamp_utc": "2026-04-11T23:02:24Z", "mode": "train", "global_step": 842, "epoch": 0.03381933566293128, "loss": 0.0297, "grad_norm": 9.17272663116455, "learning_rate": 7.451515151515152e-06, "num_tokens": 1868749.0, "completions/mean_length": 37.0, "completions/min_length": 36.0, "completions/max_length": 39.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 37.0, "completions/min_terminated_length": 36.0, "completions/max_terminated_length": 39.0, "rewards/meter/mean": 0.700322687625885, "rewards/meter/std": 0.3828321695327759, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.700322687625885, "rewards/total_composite/std": 0.3828321695327759, "reward": 0.700322687625885, "reward_std": 0.3828321397304535, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011855069547891617, "sampling/sampling_logp_difference/max": 0.964850664138794, "sampling/importance_sampling_ratio/min": 0.6239942312240601, "sampling/importance_sampling_ratio/mean": 1.0074673891067505, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.04681507125496864, "clip_ratio/low_mean": 0.009699730202555656, "clip_ratio/low_min": 0.009699730202555656, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.009699730202555656, "reward_total_mean": 0.700322687625885, "reward_meter_mean": 0.700322687625885, "reward_meter_std": 0.3828321695327759, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.700322687625885, "reward_total_composite_std": 0.3828321695327759} {"timestamp_utc": "2026-04-11T23:02:33Z", "mode": "train", "global_step": 843, "epoch": 0.03385950114471623, "loss": 0.0211, "grad_norm": 2.23907470703125, "learning_rate": 7.448484848484849e-06, "num_tokens": 1874562.0, "completions/mean_length": 452.625, "completions/min_length": 426.0, "completions/max_length": 461.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 452.625, "completions/min_terminated_length": 426.0, "completions/max_terminated_length": 461.0, "rewards/meter/mean": 0.9962563514709473, "rewards/meter/std": 0.001195572316646576, "rewards/count_adherence/mean": 0.9270833730697632, "rewards/count_adherence/std": 0.029462777078151703, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5606521368026733, "rewards/repeat_penalty/std": 0.0018446200992912054, "rewards/total_composite/mean": 0.5178769826889038, "rewards/total_composite/std": 0.01842796988785267, "reward": 0.5178769826889038, "reward_std": 0.018427973613142967, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004932690877467394, "sampling/sampling_logp_difference/max": 4.1077117919921875, "sampling/importance_sampling_ratio/min": 0.01644536294043064, "sampling/importance_sampling_ratio/mean": 0.999667763710022, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.01467959355795756, "clip_ratio/low_mean": 0.0011018531513400376, "clip_ratio/low_min": 0.0011018531513400376, "clip_ratio/high_mean": 0.0002934272342827171, "clip_ratio/high_max": 0.0002934272342827171, "clip_ratio/region_mean": 0.0013952803856227547, "reward_total_mean": 0.5178769826889038, "reward_meter_mean": 0.9962563514709473, "reward_meter_std": 0.001195572316646576, "reward_count_adherence_mean": 0.9270833730697632, "reward_count_adherence_std": 0.029462777078151703, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5606521368026733, "reward_repeat_penalty_std": 0.0018446200992912054, "reward_total_composite_mean": 0.5178769826889038, "reward_total_composite_std": 0.01842796988785267} {"timestamp_utc": "2026-04-11T23:02:39Z", "mode": "train", "global_step": 844, "epoch": 0.033899666626501186, "loss": -0.0116, "grad_norm": 2.640986919403076, "learning_rate": 7.445454545454546e-06, "num_tokens": 1877374.0, "completions/mean_length": 149.5, "completions/min_length": 144.0, "completions/max_length": 158.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 149.5, "completions/min_terminated_length": 144.0, "completions/max_terminated_length": 158.0, "rewards/meter/mean": 0.5467606782913208, "rewards/meter/std": 0.4782540798187256, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7321428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.3916763961315155, "rewards/total_composite/std": 0.34031400084495544, "reward": 0.3916763961315155, "reward_std": 0.34031397104263306, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.019425617530941963, "sampling/sampling_logp_difference/max": 3.396358013153076, "sampling/importance_sampling_ratio/min": 0.03349503502249718, "sampling/importance_sampling_ratio/mean": 1.0008183717727661, "sampling/importance_sampling_ratio/max": 1.7641745805740356, "entropy": 0.09837253391742706, "clip_ratio/low_mean": 0.011086393264122307, "clip_ratio/low_min": 0.011086393264122307, "clip_ratio/high_mean": 0.002457805967424065, "clip_ratio/high_max": 0.002457805967424065, "clip_ratio/region_mean": 0.013544199231546372, "reward_total_mean": 0.3916763961315155, "reward_meter_mean": 0.5467606782913208, "reward_meter_std": 0.4782540798187256, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7321428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_total_composite_mean": 0.3916763961315155, "reward_total_composite_std": 0.34031400084495544} {"timestamp_utc": "2026-04-11T23:02:44Z", "mode": "train", "global_step": 845, "epoch": 0.03393983210828614, "loss": 0.0112, "grad_norm": 6.265551567077637, "learning_rate": 7.442424242424243e-06, "num_tokens": 1879205.0, "completions/mean_length": 71.875, "completions/min_length": 68.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 71.875, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.522850513458252, "rewards/meter/std": 0.4167693555355072, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.5038976669311523, "rewards/total_composite/std": 0.42369261384010315, "reward": 0.5038976669311523, "reward_std": 0.42369258403778076, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03976872190833092, "sampling/sampling_logp_difference/max": 1.2724390029907227, "sampling/importance_sampling_ratio/min": 0.2801474928855896, "sampling/importance_sampling_ratio/mean": 1.0023622512817383, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.16639000456780195, "clip_ratio/low_mean": 0.01767799479421228, "clip_ratio/low_min": 0.01767799479421228, "clip_ratio/high_mean": 0.017289764247834682, "clip_ratio/high_max": 0.017289764247834682, "clip_ratio/region_mean": 0.034967759042046964, "reward_total_mean": 0.5038976669311523, "reward_meter_mean": 0.522850513458252, "reward_meter_std": 0.4167693555355072, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.5038976669311523, "reward_total_composite_std": 0.42369261384010315} {"timestamp_utc": "2026-04-11T23:02:52Z", "mode": "train", "global_step": 846, "epoch": 0.033979997590071094, "loss": -0.0053, "grad_norm": 4.094456672668457, "learning_rate": 7.439393939393939e-06, "num_tokens": 1883732.0, "completions/mean_length": 321.875, "completions/min_length": 309.0, "completions/max_length": 336.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 321.875, "completions/min_terminated_length": 309.0, "completions/max_terminated_length": 336.0, "rewards/meter/mean": 0.6540272831916809, "rewards/meter/std": 0.39356982707977295, "rewards/count_adherence/mean": 0.9861111044883728, "rewards/count_adherence/std": 0.03928370773792267, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.4677932560443878, "rewards/repeat_penalty/std": 0.22017233073711395, "rewards/total_composite/mean": 0.3497573733329773, "rewards/total_composite/std": 0.2734247148036957, "reward": 0.3497573733329773, "reward_std": 0.2734247148036957, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.007702399045228958, "sampling/sampling_logp_difference/max": 1.841371774673462, "sampling/importance_sampling_ratio/min": 0.15859971940517426, "sampling/importance_sampling_ratio/mean": 0.9985901713371277, "sampling/importance_sampling_ratio/max": 1.768311619758606, "entropy": 0.03188518015667796, "clip_ratio/low_mean": 0.0011160714784637094, "clip_ratio/low_min": 0.0011160714784637094, "clip_ratio/high_mean": 0.0058011687360703945, "clip_ratio/high_max": 0.0058011687360703945, "clip_ratio/region_mean": 0.006917240214534104, "reward_total_mean": 0.3497573733329773, "reward_meter_mean": 0.6540272831916809, "reward_meter_std": 0.39356982707977295, "reward_count_adherence_mean": 0.9861111044883728, "reward_count_adherence_std": 0.03928370773792267, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.4677932560443878, "reward_repeat_penalty_std": 0.22017233073711395, "reward_total_composite_mean": 0.3497573733329773, "reward_total_composite_std": 0.2734247148036957} {"timestamp_utc": "2026-04-11T23:02:57Z", "mode": "train", "global_step": 847, "epoch": 0.03402016307185605, "loss": 0.045, "grad_norm": 6.007223606109619, "learning_rate": 7.4363636363636375e-06, "num_tokens": 1885713.0, "completions/mean_length": 72.625, "completions/min_length": 68.0, "completions/max_length": 80.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.625, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 80.0, "rewards/meter/mean": 0.26388847827911377, "rewards/meter/std": 0.3744986653327942, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.2634860873222351, "rewards/total_composite/std": 0.3748124837875366, "reward": 0.2634860873222351, "reward_std": 0.3748124837875366, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04990806803107262, "sampling/sampling_logp_difference/max": 1.3789244890213013, "sampling/importance_sampling_ratio/min": 0.2518492639064789, "sampling/importance_sampling_ratio/mean": 1.007526159286499, "sampling/importance_sampling_ratio/max": 1.518433690071106, "entropy": 0.2849195022135973, "clip_ratio/low_mean": 0.02969917980954051, "clip_ratio/low_min": 0.02969917980954051, "clip_ratio/high_mean": 0.019930581096559763, "clip_ratio/high_max": 0.019930581096559763, "clip_ratio/region_mean": 0.04962976090610027, "reward_total_mean": 0.2634860873222351, "reward_meter_mean": 0.26388847827911377, "reward_meter_std": 0.3744986653327942, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_total_composite_mean": 0.2634860873222351, "reward_total_composite_std": 0.3748124837875366} {"timestamp_utc": "2026-04-11T23:03:02Z", "mode": "train", "global_step": 848, "epoch": 0.034060328553641, "loss": 0.0001, "grad_norm": 2.973163366317749, "learning_rate": 7.433333333333334e-06, "num_tokens": 1888010.0, "completions/mean_length": 106.125, "completions/min_length": 100.0, "completions/max_length": 114.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 106.125, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 114.0, "rewards/meter/mean": 0.6358721256256104, "rewards/meter/std": 0.32246294617652893, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.7749999761581421, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.5016140341758728, "rewards/total_composite/std": 0.2674255073070526, "reward": 0.5016140341758728, "reward_std": 0.2674255073070526, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.022296659648418427, "sampling/sampling_logp_difference/max": 1.8602898120880127, "sampling/importance_sampling_ratio/min": 0.1556275188922882, "sampling/importance_sampling_ratio/mean": 0.9971781969070435, "sampling/importance_sampling_ratio/max": 1.6591869592666626, "entropy": 0.08717937674373388, "clip_ratio/low_mean": 0.005942982388660312, "clip_ratio/low_min": 0.005942982388660312, "clip_ratio/high_mean": 0.013045326224528253, "clip_ratio/high_max": 0.013045326224528253, "clip_ratio/region_mean": 0.018988308613188565, "reward_total_mean": 0.5016140341758728, "reward_meter_mean": 0.6358721256256104, "reward_meter_std": 0.32246294617652893, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.7749999761581421, "reward_repeat_penalty_std": 0.0707106739282608, "reward_total_composite_mean": 0.5016140341758728, "reward_total_composite_std": 0.2674255073070526} {"timestamp_utc": "2026-04-11T23:03:10Z", "mode": "train", "global_step": 849, "epoch": 0.034100494035425956, "loss": 0.001, "grad_norm": 0.23050236701965332, "learning_rate": 7.430303030303031e-06, "num_tokens": 1892024.0, "completions/mean_length": 283.75, "completions/min_length": 283.0, "completions/max_length": 284.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 283.75, "completions/min_terminated_length": 283.0, "completions/max_terminated_length": 284.0, "rewards/meter/mean": 0.9965326189994812, "rewards/meter/std": 0.00021112659305799752, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.6000000238418579, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5979195833206177, "rewards/total_composite/std": 0.00012666928523685783, "reward": 0.5979195833206177, "reward_std": 0.000126663115224801, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0016167466528713703, "sampling/sampling_logp_difference/max": 0.7609295845031738, "sampling/importance_sampling_ratio/min": 0.4672318696975708, "sampling/importance_sampling_ratio/mean": 1.0001165866851807, "sampling/importance_sampling_ratio/max": 1.2486354112625122, "entropy": 0.010985135799273849, "clip_ratio/low_mean": 0.00044014083687216043, "clip_ratio/low_min": 0.00044014083687216043, "clip_ratio/high_mean": 0.00044169611646793783, "clip_ratio/high_max": 0.00044169611646793783, "clip_ratio/region_mean": 0.0008818369533400983, "reward_total_mean": 0.5979195833206177, "reward_meter_mean": 0.9965326189994812, "reward_meter_std": 0.00021112659305799752, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.6000000238418579, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.5979195833206177, "reward_total_composite_std": 0.00012666928523685783} {"timestamp_utc": "2026-04-11T23:03:15Z", "mode": "train", "global_step": 850, "epoch": 0.03414065951721091, "loss": -0.0004, "grad_norm": 7.60121488571167, "learning_rate": 7.4272727272727275e-06, "num_tokens": 1893564.0, "completions/mean_length": 35.5, "completions/min_length": 33.0, "completions/max_length": 37.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.5, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 37.0, "rewards/meter/mean": 0.9917958974838257, "rewards/meter/std": 0.008829712867736816, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8676612377166748, "rewards/total_composite/std": 0.35069888830184937, "reward": 0.8676612377166748, "reward_std": 0.350698858499527, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.04963725060224533, "sampling/sampling_logp_difference/max": 3.3897039890289307, "sampling/importance_sampling_ratio/min": 0.03371865674853325, "sampling/importance_sampling_ratio/mean": 1.0052940845489502, "sampling/importance_sampling_ratio/max": 1.9962736368179321, "entropy": 0.15574337635189295, "clip_ratio/low_mean": 0.0071428571827709675, "clip_ratio/low_min": 0.0071428571827709675, "clip_ratio/high_mean": 0.031166881788522005, "clip_ratio/high_max": 0.031166881788522005, "clip_ratio/region_mean": 0.03830973897129297, "reward_total_mean": 0.8676612377166748, "reward_meter_mean": 0.9917958974838257, "reward_meter_std": 0.008829712867736816, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_total_composite_mean": 0.8676612377166748, "reward_total_composite_std": 0.35069888830184937} {"timestamp_utc": "2026-04-11T23:04:30Z", "mode": "eval", "global_step": 850, "epoch": 0.03414065951721091, "eval_loss": NaN, "eval_runtime": 75.7065, "eval_samples_per_second": 1.374, "eval_steps_per_second": 0.172, "eval_num_tokens": 1893564.0, "eval_completions/mean_length": 209.85576923076923, "eval_completions/min_length": 62.0, "eval_completions/max_length": 402.6923076923077, "eval_completions/clipped_ratio": 0.009615384615384616, "eval_completions/mean_terminated_length": 206.50412104679987, "eval_completions/min_terminated_length": 62.0, "eval_completions/max_terminated_length": 389.9230769230769, "eval_rewards/meter/mean": 0.6430152883896461, "eval_rewards/meter/std": 0.3854568119232471, "eval_rewards/count_adherence/mean": 0.9535174920008733, "eval_rewards/count_adherence/std": 0.07147554422800358, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/repeat_penalty/mean": 0.6578034116671636, "eval_rewards/repeat_penalty/std": 0.21721924497531012, "eval_rewards/total_composite/mean": 0.4121822485556969, "eval_rewards/total_composite/std": 0.3158151931487597, "eval_reward": 0.4121822485556969, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.00518767936871602, "eval_sampling/sampling_logp_difference/max": 0.7618373265633216, "eval_sampling/importance_sampling_ratio/min": 0.4886489510536194, "eval_sampling/importance_sampling_ratio/mean": 1.001105854144463, "eval_sampling/importance_sampling_ratio/max": 1.308758864035973, "eval_entropy": 0.046050483074325785, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.4121822485556969, "eval_reward_meter_mean": 0.6430152883896461, "eval_reward_meter_std": 0.3854568119232471, "eval_reward_count_adherence_mean": 0.9535174920008733, "eval_reward_count_adherence_std": 0.07147554422800358, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_repeat_penalty_mean": 0.6578034116671636, "eval_reward_repeat_penalty_std": 0.21721924497531012, "eval_reward_total_composite_mean": 0.4121822485556969, "eval_reward_total_composite_std": 0.3158151931487597} {"timestamp_utc": "2026-04-11T23:04:39Z", "mode": "train", "global_step": 851, "epoch": 0.034180824998995864, "loss": 0.0004, "grad_norm": 2.013568162918091, "learning_rate": 7.424242424242425e-06, "num_tokens": 1896556.0, "completions/mean_length": 186.0, "completions/min_length": 178.0, "completions/max_length": 199.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 186.0, "completions/min_terminated_length": 178.0, "completions/max_terminated_length": 199.0, "rewards/meter/mean": 0.989942193031311, "rewards/meter/std": 0.007896743714809418, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/repeat_penalty/mean": 0.5795454382896423, "rewards/repeat_penalty/std": 0.16070608794689178, "rewards/total_composite/mean": 0.5742474794387817, "rewards/total_composite/std": 0.16005170345306396, "reward": 0.5742474794387817, "reward_std": 0.16005171835422516, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.011838030070066452, "sampling/sampling_logp_difference/max": 1.5690075159072876, "sampling/importance_sampling_ratio/min": 0.3101670742034912, "sampling/importance_sampling_ratio/mean": 1.0001335144042969, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.033908116864040494, "clip_ratio/low_mean": 0.0013513513840734959, "clip_ratio/low_min": 0.0013513513840734959, "clip_ratio/high_mean": 0.00752471067244187, "clip_ratio/high_max": 0.00752471067244187, "clip_ratio/region_mean": 0.008876062056515366, "reward_total_mean": 0.5742474794387817, "reward_meter_mean": 0.989942193031311, "reward_meter_std": 0.007896743714809418, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_repeat_penalty_mean": 0.5795454382896423, "reward_repeat_penalty_std": 0.16070608794689178, "reward_total_composite_mean": 0.5742474794387817, "reward_total_composite_std": 0.16005170345306396}